torch_em.data.datasets.medical.longitudinal_ct

Longitudinal-CT is a dataset of paired baseline and follow-up whole-body CT studies for longitudinal tumor lesion segmentation and tracking in metastatic melanoma.

The dataset consists of 600 CT studies (300 patients, each with a baseline and a follow-up scan acquired during systemic therapy) from the University Hospital Tubingen, with 7,182 manually segmented lesions in total (4,079 at baseline, 3,103 at follow-up). Each CT is split per body region into one or more sub-volumes. For every case and body region, the release provides: the baseline CT and its lesion mask, the follow-up CT and its lesion mask, per-timepoint lesion center-of-gravity annotations (JSON) and lesion metadata (CSV) that capture the longitudinal correspondence between baseline and follow-up lesions (eg. persistence, regression, merging, new appearance). This module exposes the basic lesion segmentation task (CT -> binary lesion mask) for both timepoints; the additional per-lesion correspondence metadata (JSON / CSV files) is not consumed by get_longitudinal_ct_dataset / get_longitudinal_ct_loader, but remains available on disk next to the images.

The dataset is located at https://fdat.uni-tuebingen.de/records/qwsry-7t837 (DOI: 10.57754/FDAT.qwsry-7t837), openly downloadable without registration. It is licensed under CC BY-NC 4.0.

This dataset is from the publication https://doi.org/10.1038/s41597-026-07466-y. Please cite it if you use this dataset in your research.

  1"""Longitudinal-CT is a dataset of paired baseline and follow-up whole-body CT studies for longitudinal
  2tumor lesion segmentation and tracking in metastatic melanoma.
  3
  4The dataset consists of 600 CT studies (300 patients, each with a baseline and a follow-up scan acquired
  5during systemic therapy) from the University Hospital Tubingen, with 7,182 manually segmented lesions in
  6total (4,079 at baseline, 3,103 at follow-up). Each CT is split per body region into one or more sub-volumes.
  7For every case and body region, the release provides: the baseline CT and its lesion mask, the follow-up CT
  8and its lesion mask, per-timepoint lesion center-of-gravity annotations (JSON) and lesion metadata (CSV) that
  9capture the longitudinal correspondence between baseline and follow-up lesions (eg. persistence, regression,
 10merging, new appearance). This module exposes the basic lesion segmentation task (CT -> binary lesion mask)
 11for both timepoints; the additional per-lesion correspondence metadata (JSON / CSV files) is not consumed by
 12`get_longitudinal_ct_dataset` / `get_longitudinal_ct_loader`, but remains available on disk next to the images.
 13
 14The dataset is located at https://fdat.uni-tuebingen.de/records/qwsry-7t837 (DOI: 10.57754/FDAT.qwsry-7t837),
 15openly downloadable without registration. It is licensed under CC BY-NC 4.0.
 16
 17This dataset is from the publication https://doi.org/10.1038/s41597-026-07466-y.
 18Please cite it if you use this dataset in your research.
 19"""
 20
 21import os
 22import re
 23from glob import glob
 24from natsort import natsorted
 25from typing import Union, Tuple, List, Literal
 26
 27from torch.utils.data import Dataset, DataLoader
 28
 29import torch_em
 30
 31from .. import util
 32
 33
 34URL = "https://fdat.uni-tuebingen.de/api/records/qwsry-7t837/files/Longitudinal-CT.zip/content"
 35CHECKSUM = "ae361c9f163ab78d9f32cf23d20cf2b8af2d23d1a59d1a575d443edea32d4f89"
 36
 37
 38def get_longitudinal_ct_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 39    """Download the Longitudinal-CT dataset.
 40
 41    NOTE: This is a large dataset (~57 GB, distributed as a single zip archive).
 42
 43    Args:
 44        path: Filepath to a folder where the data is downloaded for further processing.
 45        download: Whether to download the data if it is not present.
 46
 47    Returns:
 48        Filepath where the data is downloaded.
 49    """
 50    data_dir = os.path.join(path, "data")
 51    if os.path.exists(os.path.join(data_dir, "inputsTr")) and os.path.exists(os.path.join(data_dir, "targetsTr")):
 52        return data_dir
 53
 54    os.makedirs(path, exist_ok=True)
 55
 56    zip_path = os.path.join(path, "Longitudinal-CT.zip")
 57    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 58    util.unzip(zip_path=zip_path, dst=data_dir, remove=False)
 59
 60    return data_dir
 61
 62
 63def get_longitudinal_ct_paths(
 64    path: Union[os.PathLike, str],
 65    timepoint: Literal["baseline", "followup", "both"] = "both",
 66    download: bool = False,
 67) -> Tuple[List[str], List[str]]:
 68    """Get paths to the Longitudinal-CT data.
 69
 70    Args:
 71        path: Filepath to a folder where the data is downloaded for further processing.
 72        timepoint: The choice of timepoint. One of 'baseline', 'followup' or 'both' (default).
 73        download: Whether to download the data if it is not present.
 74
 75    Returns:
 76        List of filepaths for the image data.
 77        List of filepaths for the label data.
 78    """
 79    if timepoint not in ("baseline", "followup", "both"):
 80        raise ValueError(f"'{timepoint}' is not a valid timepoint. Choose one of 'baseline', 'followup', 'both'.")
 81
 82    data_dir = get_longitudinal_ct_data(path, download)
 83    inputs_dir = os.path.join(data_dir, "inputsTr")
 84    targets_dir = os.path.join(data_dir, "targetsTr")
 85
 86    raw_paths, label_paths = [], []
 87    if timepoint in ("baseline", "both"):
 88        for image_path in natsorted(glob(os.path.join(inputs_dir, "*_BL_img_BL_img_*.nii.gz"))):
 89            label_path = re.sub(r"_BL_img_BL_img_", "_BL_mask_BL_img_", image_path)
 90            if not os.path.exists(label_path):
 91                continue
 92            raw_paths.append(image_path)
 93            label_paths.append(label_path)
 94
 95    if timepoint in ("followup", "both"):
 96        for image_path in natsorted(glob(os.path.join(inputs_dir, "*_FU_img_FU_img_*.nii.gz"))):
 97            fname = re.sub(r"_FU_img_FU_img_", "_FU_mask_FU_img_", os.path.basename(image_path))
 98            label_path = os.path.join(targets_dir, fname)
 99            if not os.path.exists(label_path):
100                continue
101            raw_paths.append(image_path)
102            label_paths.append(label_path)
103
104    if len(raw_paths) == 0 or len(raw_paths) != len(label_paths):
105        raise RuntimeError("Something went wrong with fetching the image and label paths.")
106
107    return raw_paths, label_paths
108
109
110def get_longitudinal_ct_dataset(
111    path: Union[os.PathLike, str],
112    patch_shape: Tuple[int, ...],
113    timepoint: Literal["baseline", "followup", "both"] = "both",
114    resize_inputs: bool = False,
115    download: bool = False,
116    **kwargs
117) -> Dataset:
118    """Get the Longitudinal-CT dataset for lesion segmentation in whole-body CT.
119
120    Args:
121        path: Filepath to a folder where the data is downloaded for further processing.
122        patch_shape: The patch shape to use for training.
123        timepoint: The choice of timepoint. One of 'baseline', 'followup' or 'both' (default).
124        resize_inputs: Whether to resize inputs to the desired patch shape.
125        download: Whether to download the data if it is not present.
126        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
127
128    Returns:
129        The segmentation dataset.
130    """
131    raw_paths, label_paths = get_longitudinal_ct_paths(path, timepoint, download)
132
133    if resize_inputs:
134        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
135        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
136            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
137        )
138
139    return torch_em.default_segmentation_dataset(
140        raw_paths=raw_paths,
141        raw_key="data",
142        label_paths=label_paths,
143        label_key="data",
144        patch_shape=patch_shape,
145        is_seg_dataset=True,
146        **kwargs
147    )
148
149
150def get_longitudinal_ct_loader(
151    path: Union[os.PathLike, str],
152    batch_size: int,
153    patch_shape: Tuple[int, ...],
154    timepoint: Literal["baseline", "followup", "both"] = "both",
155    resize_inputs: bool = False,
156    download: bool = False,
157    **kwargs
158) -> DataLoader:
159    """Get the Longitudinal-CT dataloader for lesion segmentation in whole-body CT.
160
161    Args:
162        path: Filepath to a folder where the data is downloaded for further processing.
163        batch_size: The batch size for training.
164        patch_shape: The patch shape to use for training.
165        timepoint: The choice of timepoint. One of 'baseline', 'followup' or 'both' (default).
166        resize_inputs: Whether to resize inputs to the desired patch shape.
167        download: Whether to download the data if it is not present.
168        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
169
170    Returns:
171        The DataLoader.
172    """
173    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
174    dataset = get_longitudinal_ct_dataset(path, patch_shape, timepoint, resize_inputs, download, **ds_kwargs)
175    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://fdat.uni-tuebingen.de/api/records/qwsry-7t837/files/Longitudinal-CT.zip/content'
CHECKSUM = 'ae361c9f163ab78d9f32cf23d20cf2b8af2d23d1a59d1a575d443edea32d4f89'
def get_longitudinal_ct_data(path: Union[os.PathLike, str], download: bool = False) -> str:
39def get_longitudinal_ct_data(path: Union[os.PathLike, str], download: bool = False) -> str:
40    """Download the Longitudinal-CT dataset.
41
42    NOTE: This is a large dataset (~57 GB, distributed as a single zip archive).
43
44    Args:
45        path: Filepath to a folder where the data is downloaded for further processing.
46        download: Whether to download the data if it is not present.
47
48    Returns:
49        Filepath where the data is downloaded.
50    """
51    data_dir = os.path.join(path, "data")
52    if os.path.exists(os.path.join(data_dir, "inputsTr")) and os.path.exists(os.path.join(data_dir, "targetsTr")):
53        return data_dir
54
55    os.makedirs(path, exist_ok=True)
56
57    zip_path = os.path.join(path, "Longitudinal-CT.zip")
58    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
59    util.unzip(zip_path=zip_path, dst=data_dir, remove=False)
60
61    return data_dir

Download the Longitudinal-CT dataset.

NOTE: This is a large dataset (~57 GB, distributed as a single zip archive).

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_longitudinal_ct_paths( path: Union[os.PathLike, str], timepoint: Literal['baseline', 'followup', 'both'] = 'both', download: bool = False) -> Tuple[List[str], List[str]]:
 64def get_longitudinal_ct_paths(
 65    path: Union[os.PathLike, str],
 66    timepoint: Literal["baseline", "followup", "both"] = "both",
 67    download: bool = False,
 68) -> Tuple[List[str], List[str]]:
 69    """Get paths to the Longitudinal-CT data.
 70
 71    Args:
 72        path: Filepath to a folder where the data is downloaded for further processing.
 73        timepoint: The choice of timepoint. One of 'baseline', 'followup' or 'both' (default).
 74        download: Whether to download the data if it is not present.
 75
 76    Returns:
 77        List of filepaths for the image data.
 78        List of filepaths for the label data.
 79    """
 80    if timepoint not in ("baseline", "followup", "both"):
 81        raise ValueError(f"'{timepoint}' is not a valid timepoint. Choose one of 'baseline', 'followup', 'both'.")
 82
 83    data_dir = get_longitudinal_ct_data(path, download)
 84    inputs_dir = os.path.join(data_dir, "inputsTr")
 85    targets_dir = os.path.join(data_dir, "targetsTr")
 86
 87    raw_paths, label_paths = [], []
 88    if timepoint in ("baseline", "both"):
 89        for image_path in natsorted(glob(os.path.join(inputs_dir, "*_BL_img_BL_img_*.nii.gz"))):
 90            label_path = re.sub(r"_BL_img_BL_img_", "_BL_mask_BL_img_", image_path)
 91            if not os.path.exists(label_path):
 92                continue
 93            raw_paths.append(image_path)
 94            label_paths.append(label_path)
 95
 96    if timepoint in ("followup", "both"):
 97        for image_path in natsorted(glob(os.path.join(inputs_dir, "*_FU_img_FU_img_*.nii.gz"))):
 98            fname = re.sub(r"_FU_img_FU_img_", "_FU_mask_FU_img_", os.path.basename(image_path))
 99            label_path = os.path.join(targets_dir, fname)
100            if not os.path.exists(label_path):
101                continue
102            raw_paths.append(image_path)
103            label_paths.append(label_path)
104
105    if len(raw_paths) == 0 or len(raw_paths) != len(label_paths):
106        raise RuntimeError("Something went wrong with fetching the image and label paths.")
107
108    return raw_paths, label_paths

Get paths to the Longitudinal-CT data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • timepoint: The choice of timepoint. One of 'baseline', 'followup' or 'both' (default).
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_longitudinal_ct_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], timepoint: Literal['baseline', 'followup', 'both'] = 'both', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
111def get_longitudinal_ct_dataset(
112    path: Union[os.PathLike, str],
113    patch_shape: Tuple[int, ...],
114    timepoint: Literal["baseline", "followup", "both"] = "both",
115    resize_inputs: bool = False,
116    download: bool = False,
117    **kwargs
118) -> Dataset:
119    """Get the Longitudinal-CT dataset for lesion segmentation in whole-body CT.
120
121    Args:
122        path: Filepath to a folder where the data is downloaded for further processing.
123        patch_shape: The patch shape to use for training.
124        timepoint: The choice of timepoint. One of 'baseline', 'followup' or 'both' (default).
125        resize_inputs: Whether to resize inputs to the desired patch shape.
126        download: Whether to download the data if it is not present.
127        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
128
129    Returns:
130        The segmentation dataset.
131    """
132    raw_paths, label_paths = get_longitudinal_ct_paths(path, timepoint, download)
133
134    if resize_inputs:
135        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
136        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
137            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
138        )
139
140    return torch_em.default_segmentation_dataset(
141        raw_paths=raw_paths,
142        raw_key="data",
143        label_paths=label_paths,
144        label_key="data",
145        patch_shape=patch_shape,
146        is_seg_dataset=True,
147        **kwargs
148    )

Get the Longitudinal-CT dataset for lesion segmentation in whole-body CT.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • timepoint: The choice of timepoint. One of 'baseline', 'followup' or 'both' (default).
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_longitudinal_ct_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], timepoint: Literal['baseline', 'followup', 'both'] = 'both', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
151def get_longitudinal_ct_loader(
152    path: Union[os.PathLike, str],
153    batch_size: int,
154    patch_shape: Tuple[int, ...],
155    timepoint: Literal["baseline", "followup", "both"] = "both",
156    resize_inputs: bool = False,
157    download: bool = False,
158    **kwargs
159) -> DataLoader:
160    """Get the Longitudinal-CT dataloader for lesion segmentation in whole-body CT.
161
162    Args:
163        path: Filepath to a folder where the data is downloaded for further processing.
164        batch_size: The batch size for training.
165        patch_shape: The patch shape to use for training.
166        timepoint: The choice of timepoint. One of 'baseline', 'followup' or 'both' (default).
167        resize_inputs: Whether to resize inputs to the desired patch shape.
168        download: Whether to download the data if it is not present.
169        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
170
171    Returns:
172        The DataLoader.
173    """
174    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
175    dataset = get_longitudinal_ct_dataset(path, patch_shape, timepoint, resize_inputs, download, **ds_kwargs)
176    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the Longitudinal-CT dataloader for lesion segmentation in whole-body CT.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • timepoint: The choice of timepoint. One of 'baseline', 'followup' or 'both' (default).
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.