torch_em.data.datasets.medical.beamster
BEAMSTER is a dataset for segmentation of brain metastases in contrast-enhanced T1-weighted MRI.
The dataset consists of 140 retrospective contrast-enhanced T1-weighted MRI scans of patients with brain metastases who underwent stereotactic radiotherapy, with 260 metastatic lesions annotated by an experienced radiation oncologist. All MRI volumes and segmentation masks are registered to the planning CT coordinate space. The dataset is split into 'Dataset_A' (113 cases) and 'Dataset_B' (27 cases enriched with small metastases).
The dataset is located at https://doi.org/10.6084/m9.figshare.29365844 (Figshare, CC BY 4.0). The dataset is from the publication https://doi.org/10.1038/s41597-026-07777-0. Please cite it if you use this dataset for your research.
1"""BEAMSTER is a dataset for segmentation of brain metastases in contrast-enhanced T1-weighted MRI. 2 3The dataset consists of 140 retrospective contrast-enhanced T1-weighted MRI scans of patients with brain 4metastases who underwent stereotactic radiotherapy, with 260 metastatic lesions annotated by an experienced 5radiation oncologist. All MRI volumes and segmentation masks are registered to the planning CT coordinate 6space. The dataset is split into 'Dataset_A' (113 cases) and 'Dataset_B' (27 cases enriched with small 7metastases). 8 9The dataset is located at https://doi.org/10.6084/m9.figshare.29365844 (Figshare, CC BY 4.0). 10The dataset is from the publication https://doi.org/10.1038/s41597-026-07777-0. 11Please cite it if you use this dataset for your research. 12""" 13 14import os 15from glob import glob 16from natsort import natsorted 17from typing import Union, Tuple, List 18 19from torch.utils.data import Dataset, DataLoader 20 21import torch_em 22 23from .. import util 24 25 26ARTICLE_ID = 29365844 27 28 29def get_beamster_data(path: Union[os.PathLike, str], download: bool = False) -> str: 30 """Download the BEAMSTER data. 31 32 Args: 33 path: Filepath to a folder where the data is downloaded for further processing. 34 download: Whether to download the data if it is not present. 35 36 Returns: 37 Filepath where the data is downloaded. 38 """ 39 import requests 40 41 data_dir = os.path.join(path, "data") 42 marker_path = os.path.join(data_dir, ".download_complete") 43 if os.path.exists(marker_path): 44 return data_dir 45 46 if not download: 47 raise RuntimeError(f"Cannot find the data at {data_dir}, but download was set to False.") 48 49 os.makedirs(data_dir, exist_ok=True) 50 51 response = requests.get(f"https://api.figshare.com/v2/articles/{ARTICLE_ID}") 52 response.raise_for_status() 53 file_infos = [f for f in response.json()["files"] if f["name"].endswith(".nii.gz")] 54 55 for file_info in file_infos: 56 fpath = os.path.join(data_dir, file_info["name"]) 57 util.download_source(path=fpath, url=file_info["download_url"], download=download) 58 59 with open(marker_path, "w"): 60 pass 61 62 return data_dir 63 64 65def get_beamster_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 66 """Get paths to the BEAMSTER data. 67 68 Args: 69 path: Filepath to a folder where the data is downloaded for further processing. 70 download: Whether to download the data if it is not present. 71 72 Returns: 73 List of filepaths for the image data. 74 List of filepaths for the label data. 75 """ 76 data_dir = get_beamster_data(path, download) 77 78 label_paths = natsorted(glob(os.path.join(data_dir, "Dataset_*_segm.nii.gz"))) 79 raw_paths = [p.replace("_segm.nii.gz", ".nii.gz") for p in label_paths] 80 81 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 82 for raw_path in raw_paths: 83 assert os.path.exists(raw_path), raw_path 84 85 return raw_paths, label_paths 86 87 88def get_beamster_dataset( 89 path: Union[os.PathLike, str], 90 patch_shape: Tuple[int, ...], 91 resize_inputs: bool = False, 92 download: bool = False, 93 **kwargs 94) -> Dataset: 95 """Get the BEAMSTER dataset for brain metastases segmentation in MRI. 96 97 Args: 98 path: Filepath to a folder where the data is downloaded for further processing. 99 patch_shape: The patch shape to use for training. 100 resize_inputs: Whether to resize inputs to the desired patch shape. 101 download: Whether to download the data if it is not present. 102 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 103 104 Returns: 105 The segmentation dataset. 106 """ 107 raw_paths, label_paths = get_beamster_paths(path, download) 108 109 if resize_inputs: 110 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 111 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 112 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 113 ) 114 115 return torch_em.default_segmentation_dataset( 116 raw_paths=raw_paths, 117 raw_key="data", 118 label_paths=label_paths, 119 label_key="data", 120 patch_shape=patch_shape, 121 is_seg_dataset=True, 122 **kwargs 123 ) 124 125 126def get_beamster_loader( 127 path: Union[os.PathLike, str], 128 batch_size: int, 129 patch_shape: Tuple[int, ...], 130 resize_inputs: bool = False, 131 download: bool = False, 132 **kwargs 133) -> DataLoader: 134 """Get the BEAMSTER dataloader for brain metastases segmentation in MRI. 135 136 Args: 137 path: Filepath to a folder where the data is downloaded for further processing. 138 batch_size: The batch size for training. 139 patch_shape: The patch shape to use for training. 140 resize_inputs: Whether to resize inputs to the desired patch shape. 141 download: Whether to download the data if it is not present. 142 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 143 144 Returns: 145 The DataLoader. 146 """ 147 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 148 dataset = get_beamster_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 149 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
30def get_beamster_data(path: Union[os.PathLike, str], download: bool = False) -> str: 31 """Download the BEAMSTER data. 32 33 Args: 34 path: Filepath to a folder where the data is downloaded for further processing. 35 download: Whether to download the data if it is not present. 36 37 Returns: 38 Filepath where the data is downloaded. 39 """ 40 import requests 41 42 data_dir = os.path.join(path, "data") 43 marker_path = os.path.join(data_dir, ".download_complete") 44 if os.path.exists(marker_path): 45 return data_dir 46 47 if not download: 48 raise RuntimeError(f"Cannot find the data at {data_dir}, but download was set to False.") 49 50 os.makedirs(data_dir, exist_ok=True) 51 52 response = requests.get(f"https://api.figshare.com/v2/articles/{ARTICLE_ID}") 53 response.raise_for_status() 54 file_infos = [f for f in response.json()["files"] if f["name"].endswith(".nii.gz")] 55 56 for file_info in file_infos: 57 fpath = os.path.join(data_dir, file_info["name"]) 58 util.download_source(path=fpath, url=file_info["download_url"], download=download) 59 60 with open(marker_path, "w"): 61 pass 62 63 return data_dir
Download the BEAMSTER data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
66def get_beamster_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 67 """Get paths to the BEAMSTER data. 68 69 Args: 70 path: Filepath to a folder where the data is downloaded for further processing. 71 download: Whether to download the data if it is not present. 72 73 Returns: 74 List of filepaths for the image data. 75 List of filepaths for the label data. 76 """ 77 data_dir = get_beamster_data(path, download) 78 79 label_paths = natsorted(glob(os.path.join(data_dir, "Dataset_*_segm.nii.gz"))) 80 raw_paths = [p.replace("_segm.nii.gz", ".nii.gz") for p in label_paths] 81 82 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 83 for raw_path in raw_paths: 84 assert os.path.exists(raw_path), raw_path 85 86 return raw_paths, label_paths
Get paths to the BEAMSTER data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
89def get_beamster_dataset( 90 path: Union[os.PathLike, str], 91 patch_shape: Tuple[int, ...], 92 resize_inputs: bool = False, 93 download: bool = False, 94 **kwargs 95) -> Dataset: 96 """Get the BEAMSTER dataset for brain metastases segmentation in MRI. 97 98 Args: 99 path: Filepath to a folder where the data is downloaded for further processing. 100 patch_shape: The patch shape to use for training. 101 resize_inputs: Whether to resize inputs to the desired patch shape. 102 download: Whether to download the data if it is not present. 103 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 104 105 Returns: 106 The segmentation dataset. 107 """ 108 raw_paths, label_paths = get_beamster_paths(path, download) 109 110 if resize_inputs: 111 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 112 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 113 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 114 ) 115 116 return torch_em.default_segmentation_dataset( 117 raw_paths=raw_paths, 118 raw_key="data", 119 label_paths=label_paths, 120 label_key="data", 121 patch_shape=patch_shape, 122 is_seg_dataset=True, 123 **kwargs 124 )
Get the BEAMSTER dataset for brain metastases segmentation in MRI.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
127def get_beamster_loader( 128 path: Union[os.PathLike, str], 129 batch_size: int, 130 patch_shape: Tuple[int, ...], 131 resize_inputs: bool = False, 132 download: bool = False, 133 **kwargs 134) -> DataLoader: 135 """Get the BEAMSTER dataloader for brain metastases segmentation in MRI. 136 137 Args: 138 path: Filepath to a folder where the data is downloaded for further processing. 139 batch_size: The batch size for training. 140 patch_shape: The patch shape to use for training. 141 resize_inputs: Whether to resize inputs to the desired patch shape. 142 download: Whether to download the data if it is not present. 143 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 144 145 Returns: 146 The DataLoader. 147 """ 148 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 149 dataset = get_beamster_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 150 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the BEAMSTER dataloader for brain metastases segmentation in MRI.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.