torch_em.data.datasets.medical.deep_psma

The DEEP-PSMA dataset contains annotations for total tumour burden segmentation in whole-body PSMA and FDG PET/CT scans of the same patients.

The dataset consists of 100 patients with metastatic prostate cancer, each imaged with both a PSMA PET/CT scan (Ga-68 PSMA-617 or F-18 DCFPyL) and an FDG PET/CT scan, acquired prior to staging for Lu-177 PSMA therapy. This is a different dataset from torch_em.data.datasets.medical.autopet (single-tracer FDG PET/CT) and torch_em.data.datasets.medical.psma_pet_ct (single-tracer PSMA PET/CT): here every patient has a paired FDG and PSMA scan with independent total tumour burden (TTB) annotations for each tracer, curated for the DEEP-PSMA challenge (MICCAI 2026). Link: https://deep-psma.grand-challenge.org/

The data is distributed under the CC BY-NC 4.0 license and is hosted on Zenodo at https://doi.org/10.5281/zenodo.15281784. Please cite it if you use this dataset in your research.

  1"""The DEEP-PSMA dataset contains annotations for total tumour burden segmentation in whole-body
  2PSMA and FDG PET/CT scans of the same patients.
  3
  4The dataset consists of 100 patients with metastatic prostate cancer, each imaged with both a PSMA
  5PET/CT scan (Ga-68 PSMA-617 or F-18 DCFPyL) and an FDG PET/CT scan, acquired prior to staging for
  6Lu-177 PSMA therapy. This is a different dataset from `torch_em.data.datasets.medical.autopet`
  7(single-tracer FDG PET/CT) and `torch_em.data.datasets.medical.psma_pet_ct` (single-tracer PSMA
  8PET/CT): here every patient has a *paired* FDG and PSMA scan with independent total tumour burden
  9(TTB) annotations for each tracer, curated for the DEEP-PSMA challenge (MICCAI 2026).
 10Link: https://deep-psma.grand-challenge.org/
 11
 12The data is distributed under the CC BY-NC 4.0 license and is hosted on Zenodo at
 13https://doi.org/10.5281/zenodo.15281784.
 14Please cite it if you use this dataset in your research.
 15"""
 16
 17import os
 18from glob import glob
 19from natsort import natsorted
 20from typing import Tuple, Union, Literal, List
 21
 22from torch.utils.data import Dataset, DataLoader
 23
 24import torch_em
 25
 26from .. import util
 27
 28
 29URLS = {
 30    "0001-0020": "https://zenodo.org/records/15281784/files/0001-0020.zip",
 31    "0021-0040": "https://zenodo.org/records/15281784/files/0021-0040.zip",
 32    "0041-0060": "https://zenodo.org/records/15281784/files/0041-0060.zip",
 33    "0061-0080": "https://zenodo.org/records/15281784/files/0061-0080.zip",
 34    "0081-0100": "https://zenodo.org/records/15281784/files/0081-0100.zip",
 35}
 36
 37CHECKSUMS = {
 38    "0001-0020": "ae8198db5e8fc975b192fcb955ac6aa8",
 39    "0021-0040": "66e985d54ca3b6d32c139c2e3ccb530a",
 40    "0041-0060": "2668fe23ac06e12455c49009502a9496",
 41    "0061-0080": "e95924233c1811d728983514aa4627f8",
 42    "0081-0100": "f77e122734e77b90ab41d0570a1e9b3f",
 43}
 44
 45
 46def get_deep_psma_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 47    """Download the DEEP-PSMA dataset.
 48
 49    Args:
 50        path: Filepath to a folder where the data is downloaded for further processing.
 51        download: Whether to download the data if it is not present.
 52
 53    Returns:
 54        Filepath to the folder with the downloaded and extracted dataset.
 55    """
 56    data_dir = os.path.join(path, "DeepPSMA")
 57    if os.path.exists(data_dir):
 58        return data_dir
 59
 60    os.makedirs(path, exist_ok=True)
 61    for shard, url in URLS.items():
 62        zip_path = os.path.join(path, f"{shard}.zip")
 63        util.download_source(path=zip_path, url=url, download=download, checksum=None)
 64        _verify_md5_checksum(zip_path, CHECKSUMS[shard])
 65        util.unzip(zip_path=zip_path, dst=data_dir, remove=True)
 66
 67    return data_dir
 68
 69
 70def _verify_md5_checksum(path, checksum):
 71    import hashlib
 72
 73    hasher = hashlib.md5()
 74    with open(path, "rb") as f:
 75        for chunk in iter(lambda: f.read(64 * 1024 * 1024), b""):
 76            hasher.update(chunk)
 77
 78    this_checksum = hasher.hexdigest()
 79    if this_checksum != checksum:
 80        raise RuntimeError(
 81            f"The checksum of '{path}' does not match the expected checksum. "
 82            f"Expected: {checksum}, got: {this_checksum}"
 83        )
 84
 85
 86def get_deep_psma_paths(
 87    path: Union[os.PathLike, str],
 88    tracer: Literal["psma", "fdg"],
 89    modality: Literal["CT", "PET"] = "PET",
 90    download: bool = False,
 91) -> Tuple[List[str], List[str]]:
 92    """Get paths to the DEEP-PSMA data.
 93
 94    Args:
 95        path: Filepath to a folder where the data is downloaded for further processing.
 96        tracer: The choice of PET tracer. Either 'psma' or 'fdg'.
 97        modality: The choice of imaging modality. Either 'CT' or 'PET'.
 98        download: Whether to download the data if it is not present.
 99
100    Returns:
101        List of filepaths for the image data.
102        List of filepaths for the label data.
103    """
104    if tracer not in ("psma", "fdg"):
105        raise ValueError(f"'{tracer}' is not a valid tracer. Please choose one of ['psma', 'fdg'].")
106    if modality not in ("CT", "PET"):
107        raise ValueError(f"'{modality}' is not a valid modality. Please choose one of ['CT', 'PET'].")
108
109    data_dir = get_deep_psma_data(path, download)
110
111    tracer_dir = tracer.upper()
112    raw_paths = natsorted(glob(os.path.join(data_dir, "*", tracer_dir, f"{modality}.nii.gz")))
113    label_paths = natsorted(glob(os.path.join(data_dir, "*", tracer_dir, "TTB.nii.gz")))
114
115    assert len(raw_paths) > 0, f"Could not find any volumes in '{data_dir}'."
116    assert len(raw_paths) == len(label_paths)
117
118    return raw_paths, label_paths
119
120
121def get_deep_psma_dataset(
122    path: Union[os.PathLike, str],
123    patch_shape: Tuple[int, ...],
124    tracer: Literal["psma", "fdg"],
125    modality: Literal["CT", "PET"] = "PET",
126    resize_inputs: bool = False,
127    download: bool = False,
128    **kwargs
129) -> Dataset:
130    """Get the DEEP-PSMA dataset for total tumour burden segmentation in whole-body PET/CT scans.
131
132    Args:
133        path: Filepath to a folder where the data is downloaded for further processing.
134        patch_shape: The patch shape to use for training.
135        tracer: The choice of PET tracer. Either 'psma' or 'fdg'.
136        modality: The choice of imaging modality. Either 'CT' or 'PET'.
137        resize_inputs: Whether to resize the inputs.
138        download: Whether to download the data if it is not present.
139        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
140
141    Returns:
142        The segmentation dataset.
143    """
144    raw_paths, label_paths = get_deep_psma_paths(path, tracer, modality, download)
145
146    if resize_inputs:
147        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
148        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
149            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
150        )
151
152    return torch_em.default_segmentation_dataset(
153        raw_paths=raw_paths,
154        raw_key="data",
155        label_paths=label_paths,
156        label_key="data",
157        patch_shape=patch_shape,
158        **kwargs
159    )
160
161
162def get_deep_psma_loader(
163    path: Union[os.PathLike, str],
164    batch_size: int,
165    patch_shape: Tuple[int, ...],
166    tracer: Literal["psma", "fdg"],
167    modality: Literal["CT", "PET"] = "PET",
168    resize_inputs: bool = False,
169    download: bool = False,
170    **kwargs
171) -> DataLoader:
172    """Get the DEEP-PSMA dataloader for total tumour burden segmentation in whole-body PET/CT scans.
173
174    Args:
175        path: Filepath to a folder where the data is downloaded for further processing.
176        batch_size: The batch size for training.
177        patch_shape: The patch shape to use for training.
178        tracer: The choice of PET tracer. Either 'psma' or 'fdg'.
179        modality: The choice of imaging modality. Either 'CT' or 'PET'.
180        resize_inputs: Whether to resize the inputs.
181        download: Whether to download the data if it is not present.
182        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
183
184    Returns:
185        The DataLoader.
186    """
187    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
188    dataset = get_deep_psma_dataset(path, patch_shape, tracer, modality, resize_inputs, download, **ds_kwargs)
189    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URLS = {'0001-0020': 'https://zenodo.org/records/15281784/files/0001-0020.zip', '0021-0040': 'https://zenodo.org/records/15281784/files/0021-0040.zip', '0041-0060': 'https://zenodo.org/records/15281784/files/0041-0060.zip', '0061-0080': 'https://zenodo.org/records/15281784/files/0061-0080.zip', '0081-0100': 'https://zenodo.org/records/15281784/files/0081-0100.zip'}
CHECKSUMS = {'0001-0020': 'ae8198db5e8fc975b192fcb955ac6aa8', '0021-0040': '66e985d54ca3b6d32c139c2e3ccb530a', '0041-0060': '2668fe23ac06e12455c49009502a9496', '0061-0080': 'e95924233c1811d728983514aa4627f8', '0081-0100': 'f77e122734e77b90ab41d0570a1e9b3f'}
def get_deep_psma_data(path: Union[os.PathLike, str], download: bool = False) -> str:
47def get_deep_psma_data(path: Union[os.PathLike, str], download: bool = False) -> str:
48    """Download the DEEP-PSMA dataset.
49
50    Args:
51        path: Filepath to a folder where the data is downloaded for further processing.
52        download: Whether to download the data if it is not present.
53
54    Returns:
55        Filepath to the folder with the downloaded and extracted dataset.
56    """
57    data_dir = os.path.join(path, "DeepPSMA")
58    if os.path.exists(data_dir):
59        return data_dir
60
61    os.makedirs(path, exist_ok=True)
62    for shard, url in URLS.items():
63        zip_path = os.path.join(path, f"{shard}.zip")
64        util.download_source(path=zip_path, url=url, download=download, checksum=None)
65        _verify_md5_checksum(zip_path, CHECKSUMS[shard])
66        util.unzip(zip_path=zip_path, dst=data_dir, remove=True)
67
68    return data_dir

Download the DEEP-PSMA dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the folder with the downloaded and extracted dataset.

def get_deep_psma_paths( path: Union[os.PathLike, str], tracer: Literal['psma', 'fdg'], modality: Literal['CT', 'PET'] = 'PET', download: bool = False) -> Tuple[List[str], List[str]]:
 87def get_deep_psma_paths(
 88    path: Union[os.PathLike, str],
 89    tracer: Literal["psma", "fdg"],
 90    modality: Literal["CT", "PET"] = "PET",
 91    download: bool = False,
 92) -> Tuple[List[str], List[str]]:
 93    """Get paths to the DEEP-PSMA data.
 94
 95    Args:
 96        path: Filepath to a folder where the data is downloaded for further processing.
 97        tracer: The choice of PET tracer. Either 'psma' or 'fdg'.
 98        modality: The choice of imaging modality. Either 'CT' or 'PET'.
 99        download: Whether to download the data if it is not present.
100
101    Returns:
102        List of filepaths for the image data.
103        List of filepaths for the label data.
104    """
105    if tracer not in ("psma", "fdg"):
106        raise ValueError(f"'{tracer}' is not a valid tracer. Please choose one of ['psma', 'fdg'].")
107    if modality not in ("CT", "PET"):
108        raise ValueError(f"'{modality}' is not a valid modality. Please choose one of ['CT', 'PET'].")
109
110    data_dir = get_deep_psma_data(path, download)
111
112    tracer_dir = tracer.upper()
113    raw_paths = natsorted(glob(os.path.join(data_dir, "*", tracer_dir, f"{modality}.nii.gz")))
114    label_paths = natsorted(glob(os.path.join(data_dir, "*", tracer_dir, "TTB.nii.gz")))
115
116    assert len(raw_paths) > 0, f"Could not find any volumes in '{data_dir}'."
117    assert len(raw_paths) == len(label_paths)
118
119    return raw_paths, label_paths

Get paths to the DEEP-PSMA data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • tracer: The choice of PET tracer. Either 'psma' or 'fdg'.
  • modality: The choice of imaging modality. Either 'CT' or 'PET'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_deep_psma_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], tracer: Literal['psma', 'fdg'], modality: Literal['CT', 'PET'] = 'PET', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
122def get_deep_psma_dataset(
123    path: Union[os.PathLike, str],
124    patch_shape: Tuple[int, ...],
125    tracer: Literal["psma", "fdg"],
126    modality: Literal["CT", "PET"] = "PET",
127    resize_inputs: bool = False,
128    download: bool = False,
129    **kwargs
130) -> Dataset:
131    """Get the DEEP-PSMA dataset for total tumour burden segmentation in whole-body PET/CT scans.
132
133    Args:
134        path: Filepath to a folder where the data is downloaded for further processing.
135        patch_shape: The patch shape to use for training.
136        tracer: The choice of PET tracer. Either 'psma' or 'fdg'.
137        modality: The choice of imaging modality. Either 'CT' or 'PET'.
138        resize_inputs: Whether to resize the inputs.
139        download: Whether to download the data if it is not present.
140        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
141
142    Returns:
143        The segmentation dataset.
144    """
145    raw_paths, label_paths = get_deep_psma_paths(path, tracer, modality, download)
146
147    if resize_inputs:
148        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
149        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
150            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
151        )
152
153    return torch_em.default_segmentation_dataset(
154        raw_paths=raw_paths,
155        raw_key="data",
156        label_paths=label_paths,
157        label_key="data",
158        patch_shape=patch_shape,
159        **kwargs
160    )

Get the DEEP-PSMA dataset for total tumour burden segmentation in whole-body PET/CT scans.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • tracer: The choice of PET tracer. Either 'psma' or 'fdg'.
  • modality: The choice of imaging modality. Either 'CT' or 'PET'.
  • resize_inputs: Whether to resize the inputs.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_deep_psma_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], tracer: Literal['psma', 'fdg'], modality: Literal['CT', 'PET'] = 'PET', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
163def get_deep_psma_loader(
164    path: Union[os.PathLike, str],
165    batch_size: int,
166    patch_shape: Tuple[int, ...],
167    tracer: Literal["psma", "fdg"],
168    modality: Literal["CT", "PET"] = "PET",
169    resize_inputs: bool = False,
170    download: bool = False,
171    **kwargs
172) -> DataLoader:
173    """Get the DEEP-PSMA dataloader for total tumour burden segmentation in whole-body PET/CT scans.
174
175    Args:
176        path: Filepath to a folder where the data is downloaded for further processing.
177        batch_size: The batch size for training.
178        patch_shape: The patch shape to use for training.
179        tracer: The choice of PET tracer. Either 'psma' or 'fdg'.
180        modality: The choice of imaging modality. Either 'CT' or 'PET'.
181        resize_inputs: Whether to resize the inputs.
182        download: Whether to download the data if it is not present.
183        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
184
185    Returns:
186        The DataLoader.
187    """
188    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
189    dataset = get_deep_psma_dataset(path, patch_shape, tracer, modality, resize_inputs, download, **ds_kwargs)
190    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the DEEP-PSMA dataloader for total tumour burden segmentation in whole-body PET/CT scans.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • tracer: The choice of PET tracer. Either 'psma' or 'fdg'.
  • modality: The choice of imaging modality. Either 'CT' or 'PET'.
  • resize_inputs: Whether to resize the inputs.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.