torch_em.data.datasets.medical.trackrad

The TrackRAD dataset contains annotations for tumor segmentation in sagittal cine-MRI sequences, acquired during MRI-guided radiotherapy treatments.

This is the training data of the TrackRAD2025 challenge for real-time tumor tracking. It provides sagittal 2D cine-MRI sequences (a time-resolved stack of 2D frames per patient) from 585 patients, acquired at six international centers on 0.35T (ViewRay MRIdian) or 1.5T (Elekta Unity) MRI-linacs, with tumors in the thorax, abdomen and pelvis. For each labeled patient, every frame of the cine-MRI sequence has a corresponding per-pixel tumor segmentation mask (_labels.mha), and the first frame also has a separate single-frame mask (_first_label.mha).

NOTE: The challenge also ships a much larger 'unlabeled' collection (over 2.8 million frames from 477 patients) that has no segmentation masks. It is not supported here, since it cannot be used for segmentation training.

NOTE: The raw and label volumes are stored as (height, width, num_frames), i.e. the time axis is the last axis, not the first one; take this into account when choosing patch_shape.

The dataset is located at https://huggingface.co/datasets/LMUK-RADONC-PHYS-RES/TrackRAD2025 (DOI: 10.57967/hf/4539) and is distributed under the CC BY-NC 4.0 license. This dataset is from the publication https://doi.org/10.1002/mp.17964. Please cite it if you use this dataset in your research.

  1"""The TrackRAD dataset contains annotations for tumor segmentation in sagittal cine-MRI sequences,
  2acquired during MRI-guided radiotherapy treatments.
  3
  4This is the training data of the TrackRAD2025 challenge for real-time tumor tracking. It provides
  5sagittal 2D cine-MRI sequences (a time-resolved stack of 2D frames per patient) from 585 patients,
  6acquired at six international centers on 0.35T (ViewRay MRIdian) or 1.5T (Elekta Unity) MRI-linacs,
  7with tumors in the thorax, abdomen and pelvis. For each labeled patient, every frame of the cine-MRI
  8sequence has a corresponding per-pixel tumor segmentation mask (`_labels.mha`), and the first frame
  9also has a separate single-frame mask (`_first_label.mha`).
 10
 11NOTE: The challenge also ships a much larger 'unlabeled' collection (over 2.8 million frames from 477
 12patients) that has no segmentation masks. It is not supported here, since it cannot be used for
 13segmentation training.
 14
 15NOTE: The raw and label volumes are stored as (height, width, num_frames), i.e. the time axis is the
 16last axis, not the first one; take this into account when choosing `patch_shape`.
 17
 18The dataset is located at https://huggingface.co/datasets/LMUK-RADONC-PHYS-RES/TrackRAD2025
 19(DOI: 10.57967/hf/4539) and is distributed under the CC BY-NC 4.0 license.
 20This dataset is from the publication https://doi.org/10.1002/mp.17964.
 21Please cite it if you use this dataset in your research.
 22"""
 23
 24import os
 25from glob import glob
 26from natsort import natsorted
 27from typing import Union, Tuple, Literal, List
 28
 29from torch.utils.data import Dataset, DataLoader
 30
 31import torch_em
 32
 33from .. import util
 34
 35
 36REPO_ID = "LMUK-RADONC-PHYS-RES/TrackRAD2025"
 37
 38SPLITS = {
 39    "training": "trackrad2025_labeled_training_data",
 40    "pre-testing": "trackrad2025_labeled_pre-testing_data",
 41    "testing": "trackrad2025_labeled_testing_data",
 42}
 43"""Mapping from the split choice to its folder in the release."""
 44
 45
 46def get_trackrad_data(
 47    path: Union[os.PathLike, str], split: Literal["training", "pre-testing", "testing"] = "training",
 48    download: bool = False,
 49) -> str:
 50    """Download the TrackRAD dataset.
 51
 52    Args:
 53        path: Filepath to a folder where the data is downloaded for further processing.
 54        split: The choice of data split.
 55        download: Whether to download the data if it is not present.
 56
 57    Returns:
 58        Filepath where the data is downloaded.
 59    """
 60    if split not in SPLITS:
 61        raise ValueError(f"'{split}' is not a valid split. Choose from {list(SPLITS.keys())}.")
 62
 63    data_dir = os.path.join(path, SPLITS[split])
 64    if os.path.exists(data_dir):
 65        return data_dir
 66
 67    if not download:
 68        raise RuntimeError(f"Cannot find the data at {path}, but download was set to False.")
 69
 70    from huggingface_hub import snapshot_download
 71
 72    os.makedirs(path, exist_ok=True)
 73    snapshot_download(
 74        repo_id=REPO_ID, repo_type="dataset", local_dir=path, allow_patterns=f"{SPLITS[split]}/*"
 75    )
 76    return data_dir
 77
 78
 79def get_trackrad_paths(
 80    path: Union[os.PathLike, str], split: Literal["training", "pre-testing", "testing"] = "training",
 81    download: bool = False,
 82) -> Tuple[List[str], List[str]]:
 83    """Get paths to the TrackRAD data.
 84
 85    Args:
 86        path: Filepath to a folder where the data is downloaded for further processing.
 87        split: The choice of data split.
 88        download: Whether to download the data if it is not present.
 89
 90    Returns:
 91        List of filepaths for the image data.
 92        List of filepaths for the label data.
 93    """
 94    data_dir = get_trackrad_data(path=path, split=split, download=download)
 95
 96    image_paths = natsorted(glob(os.path.join(data_dir, "*", "images", "*_frames.mha")))
 97    label_paths = natsorted(glob(os.path.join(data_dir, "*", "targets", "*_labels.mha")))
 98
 99    assert len(image_paths) == len(label_paths) and len(image_paths) > 0
100
101    return image_paths, label_paths
102
103
104def get_trackrad_dataset(
105    path: Union[os.PathLike, str],
106    patch_shape: Tuple[int, ...],
107    split: Literal["training", "pre-testing", "testing"] = "training",
108    resize_inputs: bool = False,
109    download: bool = False,
110    **kwargs
111) -> Dataset:
112    """Get the TrackRAD dataset for tumor segmentation in cine-MRI sequences.
113
114    Args:
115        path: Filepath to a folder where the data is downloaded for further processing.
116        patch_shape: The patch shape to use for training.
117        split: The choice of data split.
118        resize_inputs: Whether to resize inputs to the desired patch shape.
119        download: Whether to download the data if it is not present.
120        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
121
122    Returns:
123        The segmentation dataset.
124    """
125    image_paths, label_paths = get_trackrad_paths(path=path, split=split, download=download)
126
127    if resize_inputs:
128        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
129        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
130            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
131        )
132
133    return torch_em.default_segmentation_dataset(
134        raw_paths=image_paths,
135        raw_key=None,
136        label_paths=label_paths,
137        label_key=None,
138        patch_shape=patch_shape,
139        **kwargs
140    )
141
142
143def get_trackrad_loader(
144    path: Union[os.PathLike, str],
145    batch_size: int,
146    patch_shape: Tuple[int, ...],
147    split: Literal["training", "pre-testing", "testing"] = "training",
148    resize_inputs: bool = False,
149    download: bool = False,
150    **kwargs
151) -> DataLoader:
152    """Get the TrackRAD dataloader for tumor segmentation in cine-MRI sequences.
153
154    Args:
155        path: Filepath to a folder where the data is downloaded for further processing.
156        batch_size: The batch size for training.
157        patch_shape: The patch shape to use for training.
158        split: The choice of data split.
159        resize_inputs: Whether to resize inputs to the desired patch shape.
160        download: Whether to download the data if it is not present.
161        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
162
163    Returns:
164        The DataLoader.
165    """
166    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
167    dataset = get_trackrad_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs)
168    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
REPO_ID = 'LMUK-RADONC-PHYS-RES/TrackRAD2025'
SPLITS = {'training': 'trackrad2025_labeled_training_data', 'pre-testing': 'trackrad2025_labeled_pre-testing_data', 'testing': 'trackrad2025_labeled_testing_data'}

Mapping from the split choice to its folder in the release.

def get_trackrad_data( path: Union[os.PathLike, str], split: Literal['training', 'pre-testing', 'testing'] = 'training', download: bool = False) -> str:
47def get_trackrad_data(
48    path: Union[os.PathLike, str], split: Literal["training", "pre-testing", "testing"] = "training",
49    download: bool = False,
50) -> str:
51    """Download the TrackRAD dataset.
52
53    Args:
54        path: Filepath to a folder where the data is downloaded for further processing.
55        split: The choice of data split.
56        download: Whether to download the data if it is not present.
57
58    Returns:
59        Filepath where the data is downloaded.
60    """
61    if split not in SPLITS:
62        raise ValueError(f"'{split}' is not a valid split. Choose from {list(SPLITS.keys())}.")
63
64    data_dir = os.path.join(path, SPLITS[split])
65    if os.path.exists(data_dir):
66        return data_dir
67
68    if not download:
69        raise RuntimeError(f"Cannot find the data at {path}, but download was set to False.")
70
71    from huggingface_hub import snapshot_download
72
73    os.makedirs(path, exist_ok=True)
74    snapshot_download(
75        repo_id=REPO_ID, repo_type="dataset", local_dir=path, allow_patterns=f"{SPLITS[split]}/*"
76    )
77    return data_dir

Download the TrackRAD dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The choice of data split.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_trackrad_paths( path: Union[os.PathLike, str], split: Literal['training', 'pre-testing', 'testing'] = 'training', download: bool = False) -> Tuple[List[str], List[str]]:
 80def get_trackrad_paths(
 81    path: Union[os.PathLike, str], split: Literal["training", "pre-testing", "testing"] = "training",
 82    download: bool = False,
 83) -> Tuple[List[str], List[str]]:
 84    """Get paths to the TrackRAD data.
 85
 86    Args:
 87        path: Filepath to a folder where the data is downloaded for further processing.
 88        split: The choice of data split.
 89        download: Whether to download the data if it is not present.
 90
 91    Returns:
 92        List of filepaths for the image data.
 93        List of filepaths for the label data.
 94    """
 95    data_dir = get_trackrad_data(path=path, split=split, download=download)
 96
 97    image_paths = natsorted(glob(os.path.join(data_dir, "*", "images", "*_frames.mha")))
 98    label_paths = natsorted(glob(os.path.join(data_dir, "*", "targets", "*_labels.mha")))
 99
100    assert len(image_paths) == len(label_paths) and len(image_paths) > 0
101
102    return image_paths, label_paths

Get paths to the TrackRAD data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The choice of data split.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_trackrad_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], split: Literal['training', 'pre-testing', 'testing'] = 'training', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
105def get_trackrad_dataset(
106    path: Union[os.PathLike, str],
107    patch_shape: Tuple[int, ...],
108    split: Literal["training", "pre-testing", "testing"] = "training",
109    resize_inputs: bool = False,
110    download: bool = False,
111    **kwargs
112) -> Dataset:
113    """Get the TrackRAD dataset for tumor segmentation in cine-MRI sequences.
114
115    Args:
116        path: Filepath to a folder where the data is downloaded for further processing.
117        patch_shape: The patch shape to use for training.
118        split: The choice of data split.
119        resize_inputs: Whether to resize inputs to the desired patch shape.
120        download: Whether to download the data if it is not present.
121        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
122
123    Returns:
124        The segmentation dataset.
125    """
126    image_paths, label_paths = get_trackrad_paths(path=path, split=split, download=download)
127
128    if resize_inputs:
129        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
130        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
131            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
132        )
133
134    return torch_em.default_segmentation_dataset(
135        raw_paths=image_paths,
136        raw_key=None,
137        label_paths=label_paths,
138        label_key=None,
139        patch_shape=patch_shape,
140        **kwargs
141    )

Get the TrackRAD dataset for tumor segmentation in cine-MRI sequences.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_trackrad_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], split: Literal['training', 'pre-testing', 'testing'] = 'training', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
144def get_trackrad_loader(
145    path: Union[os.PathLike, str],
146    batch_size: int,
147    patch_shape: Tuple[int, ...],
148    split: Literal["training", "pre-testing", "testing"] = "training",
149    resize_inputs: bool = False,
150    download: bool = False,
151    **kwargs
152) -> DataLoader:
153    """Get the TrackRAD dataloader for tumor segmentation in cine-MRI sequences.
154
155    Args:
156        path: Filepath to a folder where the data is downloaded for further processing.
157        batch_size: The batch size for training.
158        patch_shape: The patch shape to use for training.
159        split: The choice of data split.
160        resize_inputs: Whether to resize inputs to the desired patch shape.
161        download: Whether to download the data if it is not present.
162        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
163
164    Returns:
165        The DataLoader.
166    """
167    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
168    dataset = get_trackrad_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs)
169    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the TrackRAD dataloader for tumor segmentation in cine-MRI sequences.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.