torch_em.data.datasets.medical.recist_ct

The RECIST CT dataset contains annotations for instance segmentation of tumors, metastases and lymph nodes in CT scans, together with the corresponding RECIST 1.1 diameter measurements.

The dataset consists of 1,246 manually segmented lesions from 58 CT scans of 22 cancer patients treated at the Clinical Hospital of the University of Chile (HCUCH). The data is split by anatomical region ('abdomen', 'thorax') and by 'train' / 'test' subset.

This dataset is from the publication https://doi.org/10.1038/s41597-026-06597-6. Please cite it if you use this dataset in your research.

  1"""The RECIST CT dataset contains annotations for instance segmentation of tumors, metastases and
  2lymph nodes in CT scans, together with the corresponding RECIST 1.1 diameter measurements.
  3
  4The dataset consists of 1,246 manually segmented lesions from 58 CT scans of 22 cancer patients treated
  5at the Clinical Hospital of the University of Chile (HCUCH). The data is split by anatomical region
  6('abdomen', 'thorax') and by 'train' / 'test' subset.
  7
  8This dataset is from the publication https://doi.org/10.1038/s41597-026-06597-6.
  9Please cite it if you use this dataset in your research.
 10"""
 11
 12import os
 13from glob import glob
 14from natsort import natsorted
 15from typing import Union, Tuple, List, Literal, Optional
 16
 17from torch.utils.data import Dataset, DataLoader
 18
 19import torch_em
 20
 21from .. import util
 22
 23
 24URL = "https://zenodo.org/records/17788162/files/final-formatted.zip"
 25CHECKSUM = "39e2fc8a34ef7f519617901e4683462a75c4d97e5c46eedd50d3b35b22b96be5"
 26
 27REGIONS = ["abdomen", "thorax"]
 28SPLITS = ["train", "test"]
 29
 30
 31def get_recist_ct_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 32    """Download the RECIST CT dataset.
 33
 34    Args:
 35        path: Filepath to a folder where the data is downloaded for further processing.
 36        download: Whether to download the data if it is not present.
 37
 38    Returns:
 39        Filepath where the data is stored.
 40    """
 41    data_dir = os.path.join(path, "final-formatted")
 42    if os.path.exists(data_dir):
 43        return data_dir
 44
 45    os.makedirs(path, exist_ok=True)
 46
 47    zip_path = os.path.join(path, "final-formatted.zip")
 48    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 49    util.unzip(zip_path=zip_path, dst=path)
 50
 51    return data_dir
 52
 53
 54def get_recist_ct_paths(
 55    path: Union[os.PathLike, str],
 56    region: Optional[Literal["abdomen", "thorax"]] = None,
 57    split: Optional[Literal["train", "test"]] = None,
 58    download: bool = False,
 59) -> Tuple[List[str], List[str]]:
 60    """Get paths to the RECIST CT data.
 61
 62    Args:
 63        path: Filepath to a folder where the data is downloaded for further processing.
 64        region: The choice of anatomical region.
 65        split: The choice of data split.
 66        download: Whether to download the data if it is not present.
 67
 68    Returns:
 69        List of filepaths for the image data.
 70        List of filepaths for the label data.
 71    """
 72    data_dir = get_recist_ct_data(path, download)
 73
 74    if region is None:
 75        regions = REGIONS
 76    else:
 77        assert region in REGIONS, f"'{region}' is not a valid region."
 78        regions = [region]
 79
 80    if split is None:
 81        splits = SPLITS
 82    else:
 83        assert split in SPLITS, f"'{split}' is not a valid split."
 84        splits = [split]
 85
 86    # The original filenames combine the patient id with the DICOM Series Instance UID, which contains many
 87    # dots. 'elf.io.open_file' only recognizes '.nii.gz' files that have exactly two suffixes, so symlinks
 88    # with the extra dots replaced by underscores are created here to make the files resolvable.
 89    clean_dir = os.path.join(path, "clean")
 90
 91    image_paths, gt_paths = [], []
 92    for _region in regions:
 93        for _split in splits:
 94            base_dir = os.path.join(data_dir, "images", _region, _split)
 95            raw_image_paths = natsorted(glob(os.path.join(base_dir, "images", "*.nii.gz")))
 96            raw_gt_paths = natsorted(glob(os.path.join(base_dir, "masks", "*.nii.gz")))
 97
 98            for raw_path, sub_dir, out_paths in [
 99                (raw_image_paths, "images", image_paths), (raw_gt_paths, "masks", gt_paths)
100            ]:
101                out_dir = os.path.join(clean_dir, _region, _split, sub_dir)
102                os.makedirs(out_dir, exist_ok=True)
103                for orig_path in raw_path:
104                    clean_name = os.path.basename(orig_path)[:-len(".nii.gz")].replace(".", "_") + ".nii.gz"
105                    clean_path = os.path.join(out_dir, clean_name)
106                    if not os.path.exists(clean_path):
107                        os.symlink(os.path.abspath(orig_path), clean_path)
108                    out_paths.append(clean_path)
109
110    assert len(image_paths) > 0 and len(image_paths) == len(gt_paths), \
111        f"Could not find a matching number of images and labels in '{data_dir}'."
112
113    return image_paths, gt_paths
114
115
116def get_recist_ct_dataset(
117    path: Union[os.PathLike, str],
118    patch_shape: Tuple[int, ...],
119    region: Optional[Literal["abdomen", "thorax"]] = None,
120    split: Optional[Literal["train", "test"]] = None,
121    resize_inputs: bool = False,
122    download: bool = False,
123    **kwargs
124) -> Dataset:
125    """Get the RECIST CT dataset for instance segmentation of tumors, metastases and lymph nodes.
126
127    Args:
128        path: Filepath to a folder where the data is downloaded for further processing.
129        patch_shape: The patch shape to use for training.
130        region: The choice of anatomical region.
131        split: The choice of data split.
132        resize_inputs: Whether to resize the inputs to the patch shape.
133        download: Whether to download the data if it is not present.
134        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
135
136    Returns:
137        The segmentation dataset.
138    """
139    image_paths, gt_paths = get_recist_ct_paths(path, region, split, download)
140
141    if resize_inputs:
142        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
143        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
144            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
145        )
146
147    return torch_em.default_segmentation_dataset(
148        raw_paths=image_paths,
149        raw_key="data",
150        label_paths=gt_paths,
151        label_key="data",
152        patch_shape=patch_shape,
153        is_seg_dataset=True,
154        **kwargs
155    )
156
157
158def get_recist_ct_loader(
159    path: Union[os.PathLike, str],
160    batch_size: int,
161    patch_shape: Tuple[int, ...],
162    region: Optional[Literal["abdomen", "thorax"]] = None,
163    split: Optional[Literal["train", "test"]] = None,
164    resize_inputs: bool = False,
165    download: bool = False,
166    **kwargs
167) -> DataLoader:
168    """Get the RECIST CT dataloader for instance segmentation of tumors, metastases and lymph nodes.
169
170    Args:
171        path: Filepath to a folder where the data is downloaded for further processing.
172        batch_size: The batch size for training.
173        patch_shape: The patch shape to use for training.
174        region: The choice of anatomical region.
175        split: The choice of data split.
176        resize_inputs: Whether to resize the inputs to the patch shape.
177        download: Whether to download the data if it is not present.
178        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or the PyTorch DataLoader.
179
180    Returns:
181        The DataLoader.
182    """
183    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
184    dataset = get_recist_ct_dataset(path, patch_shape, region, split, resize_inputs, download, **ds_kwargs)
185    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/records/17788162/files/final-formatted.zip'
CHECKSUM = '39e2fc8a34ef7f519617901e4683462a75c4d97e5c46eedd50d3b35b22b96be5'
REGIONS = ['abdomen', 'thorax']
SPLITS = ['train', 'test']
def get_recist_ct_data(path: Union[os.PathLike, str], download: bool = False) -> str:
32def get_recist_ct_data(path: Union[os.PathLike, str], download: bool = False) -> str:
33    """Download the RECIST CT dataset.
34
35    Args:
36        path: Filepath to a folder where the data is downloaded for further processing.
37        download: Whether to download the data if it is not present.
38
39    Returns:
40        Filepath where the data is stored.
41    """
42    data_dir = os.path.join(path, "final-formatted")
43    if os.path.exists(data_dir):
44        return data_dir
45
46    os.makedirs(path, exist_ok=True)
47
48    zip_path = os.path.join(path, "final-formatted.zip")
49    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
50    util.unzip(zip_path=zip_path, dst=path)
51
52    return data_dir

Download the RECIST CT dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is stored.

def get_recist_ct_paths( path: Union[os.PathLike, str], region: Optional[Literal['abdomen', 'thorax']] = None, split: Optional[Literal['train', 'test']] = None, download: bool = False) -> Tuple[List[str], List[str]]:
 55def get_recist_ct_paths(
 56    path: Union[os.PathLike, str],
 57    region: Optional[Literal["abdomen", "thorax"]] = None,
 58    split: Optional[Literal["train", "test"]] = None,
 59    download: bool = False,
 60) -> Tuple[List[str], List[str]]:
 61    """Get paths to the RECIST CT data.
 62
 63    Args:
 64        path: Filepath to a folder where the data is downloaded for further processing.
 65        region: The choice of anatomical region.
 66        split: The choice of data split.
 67        download: Whether to download the data if it is not present.
 68
 69    Returns:
 70        List of filepaths for the image data.
 71        List of filepaths for the label data.
 72    """
 73    data_dir = get_recist_ct_data(path, download)
 74
 75    if region is None:
 76        regions = REGIONS
 77    else:
 78        assert region in REGIONS, f"'{region}' is not a valid region."
 79        regions = [region]
 80
 81    if split is None:
 82        splits = SPLITS
 83    else:
 84        assert split in SPLITS, f"'{split}' is not a valid split."
 85        splits = [split]
 86
 87    # The original filenames combine the patient id with the DICOM Series Instance UID, which contains many
 88    # dots. 'elf.io.open_file' only recognizes '.nii.gz' files that have exactly two suffixes, so symlinks
 89    # with the extra dots replaced by underscores are created here to make the files resolvable.
 90    clean_dir = os.path.join(path, "clean")
 91
 92    image_paths, gt_paths = [], []
 93    for _region in regions:
 94        for _split in splits:
 95            base_dir = os.path.join(data_dir, "images", _region, _split)
 96            raw_image_paths = natsorted(glob(os.path.join(base_dir, "images", "*.nii.gz")))
 97            raw_gt_paths = natsorted(glob(os.path.join(base_dir, "masks", "*.nii.gz")))
 98
 99            for raw_path, sub_dir, out_paths in [
100                (raw_image_paths, "images", image_paths), (raw_gt_paths, "masks", gt_paths)
101            ]:
102                out_dir = os.path.join(clean_dir, _region, _split, sub_dir)
103                os.makedirs(out_dir, exist_ok=True)
104                for orig_path in raw_path:
105                    clean_name = os.path.basename(orig_path)[:-len(".nii.gz")].replace(".", "_") + ".nii.gz"
106                    clean_path = os.path.join(out_dir, clean_name)
107                    if not os.path.exists(clean_path):
108                        os.symlink(os.path.abspath(orig_path), clean_path)
109                    out_paths.append(clean_path)
110
111    assert len(image_paths) > 0 and len(image_paths) == len(gt_paths), \
112        f"Could not find a matching number of images and labels in '{data_dir}'."
113
114    return image_paths, gt_paths

Get paths to the RECIST CT data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • region: The choice of anatomical region.
  • split: The choice of data split.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_recist_ct_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], region: Optional[Literal['abdomen', 'thorax']] = None, split: Optional[Literal['train', 'test']] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
117def get_recist_ct_dataset(
118    path: Union[os.PathLike, str],
119    patch_shape: Tuple[int, ...],
120    region: Optional[Literal["abdomen", "thorax"]] = None,
121    split: Optional[Literal["train", "test"]] = None,
122    resize_inputs: bool = False,
123    download: bool = False,
124    **kwargs
125) -> Dataset:
126    """Get the RECIST CT dataset for instance segmentation of tumors, metastases and lymph nodes.
127
128    Args:
129        path: Filepath to a folder where the data is downloaded for further processing.
130        patch_shape: The patch shape to use for training.
131        region: The choice of anatomical region.
132        split: The choice of data split.
133        resize_inputs: Whether to resize the inputs to the patch shape.
134        download: Whether to download the data if it is not present.
135        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
136
137    Returns:
138        The segmentation dataset.
139    """
140    image_paths, gt_paths = get_recist_ct_paths(path, region, split, download)
141
142    if resize_inputs:
143        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
144        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
145            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
146        )
147
148    return torch_em.default_segmentation_dataset(
149        raw_paths=image_paths,
150        raw_key="data",
151        label_paths=gt_paths,
152        label_key="data",
153        patch_shape=patch_shape,
154        is_seg_dataset=True,
155        **kwargs
156    )

Get the RECIST CT dataset for instance segmentation of tumors, metastases and lymph nodes.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • region: The choice of anatomical region.
  • split: The choice of data split.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_recist_ct_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], region: Optional[Literal['abdomen', 'thorax']] = None, split: Optional[Literal['train', 'test']] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
159def get_recist_ct_loader(
160    path: Union[os.PathLike, str],
161    batch_size: int,
162    patch_shape: Tuple[int, ...],
163    region: Optional[Literal["abdomen", "thorax"]] = None,
164    split: Optional[Literal["train", "test"]] = None,
165    resize_inputs: bool = False,
166    download: bool = False,
167    **kwargs
168) -> DataLoader:
169    """Get the RECIST CT dataloader for instance segmentation of tumors, metastases and lymph nodes.
170
171    Args:
172        path: Filepath to a folder where the data is downloaded for further processing.
173        batch_size: The batch size for training.
174        patch_shape: The patch shape to use for training.
175        region: The choice of anatomical region.
176        split: The choice of data split.
177        resize_inputs: Whether to resize the inputs to the patch shape.
178        download: Whether to download the data if it is not present.
179        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or the PyTorch DataLoader.
180
181    Returns:
182        The DataLoader.
183    """
184    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
185    dataset = get_recist_ct_dataset(path, patch_shape, region, split, resize_inputs, download, **ds_kwargs)
186    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the RECIST CT dataloader for instance segmentation of tumors, metastases and lymph nodes.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • region: The choice of anatomical region.
  • split: The choice of data split.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or the PyTorch DataLoader.
Returns:

The DataLoader.