torch_em.data.datasets.medical.beamster

BEAMSTER is a dataset for segmentation of brain metastases in contrast-enhanced T1-weighted MRI.

The dataset consists of 140 retrospective contrast-enhanced T1-weighted MRI scans of patients with brain metastases who underwent stereotactic radiotherapy, with 260 metastatic lesions annotated by an experienced radiation oncologist. All MRI volumes and segmentation masks are registered to the planning CT coordinate space. The dataset is split into 'Dataset_A' (113 cases) and 'Dataset_B' (27 cases enriched with small metastases).

The dataset is located at https://doi.org/10.6084/m9.figshare.29365844 (Figshare, CC BY 4.0). The dataset is from the publication https://doi.org/10.1038/s41597-026-07777-0. Please cite it if you use this dataset for your research.

  1"""BEAMSTER is a dataset for segmentation of brain metastases in contrast-enhanced T1-weighted MRI.
  2
  3The dataset consists of 140 retrospective contrast-enhanced T1-weighted MRI scans of patients with brain
  4metastases who underwent stereotactic radiotherapy, with 260 metastatic lesions annotated by an experienced
  5radiation oncologist. All MRI volumes and segmentation masks are registered to the planning CT coordinate
  6space. The dataset is split into 'Dataset_A' (113 cases) and 'Dataset_B' (27 cases enriched with small
  7metastases).
  8
  9The dataset is located at https://doi.org/10.6084/m9.figshare.29365844 (Figshare, CC BY 4.0).
 10The dataset is from the publication https://doi.org/10.1038/s41597-026-07777-0.
 11Please cite it if you use this dataset for your research.
 12"""
 13
 14import os
 15from glob import glob
 16from natsort import natsorted
 17from typing import Union, Tuple, List
 18
 19from torch.utils.data import Dataset, DataLoader
 20
 21import torch_em
 22
 23from .. import util
 24
 25
 26ARTICLE_ID = 29365844
 27
 28
 29def get_beamster_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 30    """Download the BEAMSTER data.
 31
 32    Args:
 33        path: Filepath to a folder where the data is downloaded for further processing.
 34        download: Whether to download the data if it is not present.
 35
 36    Returns:
 37        Filepath where the data is downloaded.
 38    """
 39    import requests
 40
 41    data_dir = os.path.join(path, "data")
 42    marker_path = os.path.join(data_dir, ".download_complete")
 43    if os.path.exists(marker_path):
 44        return data_dir
 45
 46    if not download:
 47        raise RuntimeError(f"Cannot find the data at {data_dir}, but download was set to False.")
 48
 49    os.makedirs(data_dir, exist_ok=True)
 50
 51    response = requests.get(f"https://api.figshare.com/v2/articles/{ARTICLE_ID}")
 52    response.raise_for_status()
 53    file_infos = [f for f in response.json()["files"] if f["name"].endswith(".nii.gz")]
 54
 55    for file_info in file_infos:
 56        fpath = os.path.join(data_dir, file_info["name"])
 57        util.download_source(path=fpath, url=file_info["download_url"], download=download)
 58
 59    with open(marker_path, "w"):
 60        pass
 61
 62    return data_dir
 63
 64
 65def get_beamster_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 66    """Get paths to the BEAMSTER data.
 67
 68    Args:
 69        path: Filepath to a folder where the data is downloaded for further processing.
 70        download: Whether to download the data if it is not present.
 71
 72    Returns:
 73        List of filepaths for the image data.
 74        List of filepaths for the label data.
 75    """
 76    data_dir = get_beamster_data(path, download)
 77
 78    label_paths = natsorted(glob(os.path.join(data_dir, "Dataset_*_segm.nii.gz")))
 79    raw_paths = [p.replace("_segm.nii.gz", ".nii.gz") for p in label_paths]
 80
 81    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
 82    for raw_path in raw_paths:
 83        assert os.path.exists(raw_path), raw_path
 84
 85    return raw_paths, label_paths
 86
 87
 88def get_beamster_dataset(
 89    path: Union[os.PathLike, str],
 90    patch_shape: Tuple[int, ...],
 91    resize_inputs: bool = False,
 92    download: bool = False,
 93    **kwargs
 94) -> Dataset:
 95    """Get the BEAMSTER dataset for brain metastases segmentation in MRI.
 96
 97    Args:
 98        path: Filepath to a folder where the data is downloaded for further processing.
 99        patch_shape: The patch shape to use for training.
100        resize_inputs: Whether to resize inputs to the desired patch shape.
101        download: Whether to download the data if it is not present.
102        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
103
104    Returns:
105        The segmentation dataset.
106    """
107    raw_paths, label_paths = get_beamster_paths(path, download)
108
109    if resize_inputs:
110        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
111        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
112            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
113        )
114
115    return torch_em.default_segmentation_dataset(
116        raw_paths=raw_paths,
117        raw_key="data",
118        label_paths=label_paths,
119        label_key="data",
120        patch_shape=patch_shape,
121        is_seg_dataset=True,
122        **kwargs
123    )
124
125
126def get_beamster_loader(
127    path: Union[os.PathLike, str],
128    batch_size: int,
129    patch_shape: Tuple[int, ...],
130    resize_inputs: bool = False,
131    download: bool = False,
132    **kwargs
133) -> DataLoader:
134    """Get the BEAMSTER dataloader for brain metastases segmentation in MRI.
135
136    Args:
137        path: Filepath to a folder where the data is downloaded for further processing.
138        batch_size: The batch size for training.
139        patch_shape: The patch shape to use for training.
140        resize_inputs: Whether to resize inputs to the desired patch shape.
141        download: Whether to download the data if it is not present.
142        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
143
144    Returns:
145        The DataLoader.
146    """
147    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
148    dataset = get_beamster_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
149    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
ARTICLE_ID = 29365844
def get_beamster_data(path: Union[os.PathLike, str], download: bool = False) -> str:
30def get_beamster_data(path: Union[os.PathLike, str], download: bool = False) -> str:
31    """Download the BEAMSTER data.
32
33    Args:
34        path: Filepath to a folder where the data is downloaded for further processing.
35        download: Whether to download the data if it is not present.
36
37    Returns:
38        Filepath where the data is downloaded.
39    """
40    import requests
41
42    data_dir = os.path.join(path, "data")
43    marker_path = os.path.join(data_dir, ".download_complete")
44    if os.path.exists(marker_path):
45        return data_dir
46
47    if not download:
48        raise RuntimeError(f"Cannot find the data at {data_dir}, but download was set to False.")
49
50    os.makedirs(data_dir, exist_ok=True)
51
52    response = requests.get(f"https://api.figshare.com/v2/articles/{ARTICLE_ID}")
53    response.raise_for_status()
54    file_infos = [f for f in response.json()["files"] if f["name"].endswith(".nii.gz")]
55
56    for file_info in file_infos:
57        fpath = os.path.join(data_dir, file_info["name"])
58        util.download_source(path=fpath, url=file_info["download_url"], download=download)
59
60    with open(marker_path, "w"):
61        pass
62
63    return data_dir

Download the BEAMSTER data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_beamster_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
66def get_beamster_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
67    """Get paths to the BEAMSTER data.
68
69    Args:
70        path: Filepath to a folder where the data is downloaded for further processing.
71        download: Whether to download the data if it is not present.
72
73    Returns:
74        List of filepaths for the image data.
75        List of filepaths for the label data.
76    """
77    data_dir = get_beamster_data(path, download)
78
79    label_paths = natsorted(glob(os.path.join(data_dir, "Dataset_*_segm.nii.gz")))
80    raw_paths = [p.replace("_segm.nii.gz", ".nii.gz") for p in label_paths]
81
82    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
83    for raw_path in raw_paths:
84        assert os.path.exists(raw_path), raw_path
85
86    return raw_paths, label_paths

Get paths to the BEAMSTER data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_beamster_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
 89def get_beamster_dataset(
 90    path: Union[os.PathLike, str],
 91    patch_shape: Tuple[int, ...],
 92    resize_inputs: bool = False,
 93    download: bool = False,
 94    **kwargs
 95) -> Dataset:
 96    """Get the BEAMSTER dataset for brain metastases segmentation in MRI.
 97
 98    Args:
 99        path: Filepath to a folder where the data is downloaded for further processing.
100        patch_shape: The patch shape to use for training.
101        resize_inputs: Whether to resize inputs to the desired patch shape.
102        download: Whether to download the data if it is not present.
103        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
104
105    Returns:
106        The segmentation dataset.
107    """
108    raw_paths, label_paths = get_beamster_paths(path, download)
109
110    if resize_inputs:
111        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
112        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
113            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
114        )
115
116    return torch_em.default_segmentation_dataset(
117        raw_paths=raw_paths,
118        raw_key="data",
119        label_paths=label_paths,
120        label_key="data",
121        patch_shape=patch_shape,
122        is_seg_dataset=True,
123        **kwargs
124    )

Get the BEAMSTER dataset for brain metastases segmentation in MRI.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_beamster_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
127def get_beamster_loader(
128    path: Union[os.PathLike, str],
129    batch_size: int,
130    patch_shape: Tuple[int, ...],
131    resize_inputs: bool = False,
132    download: bool = False,
133    **kwargs
134) -> DataLoader:
135    """Get the BEAMSTER dataloader for brain metastases segmentation in MRI.
136
137    Args:
138        path: Filepath to a folder where the data is downloaded for further processing.
139        batch_size: The batch size for training.
140        patch_shape: The patch shape to use for training.
141        resize_inputs: Whether to resize inputs to the desired patch shape.
142        download: Whether to download the data if it is not present.
143        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
144
145    Returns:
146        The DataLoader.
147    """
148    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
149    dataset = get_beamster_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
150    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the BEAMSTER dataloader for brain metastases segmentation in MRI.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.