torch_em.data.datasets.medical.recist_ct
The RECIST CT dataset contains annotations for instance segmentation of tumors, metastases and lymph nodes in CT scans, together with the corresponding RECIST 1.1 diameter measurements.
The dataset consists of 1,246 manually segmented lesions from 58 CT scans of 22 cancer patients treated at the Clinical Hospital of the University of Chile (HCUCH). The data is split by anatomical region ('abdomen', 'thorax') and by 'train' / 'test' subset.
This dataset is from the publication https://doi.org/10.1038/s41597-026-06597-6. Please cite it if you use this dataset in your research.
1"""The RECIST CT dataset contains annotations for instance segmentation of tumors, metastases and 2lymph nodes in CT scans, together with the corresponding RECIST 1.1 diameter measurements. 3 4The dataset consists of 1,246 manually segmented lesions from 58 CT scans of 22 cancer patients treated 5at the Clinical Hospital of the University of Chile (HCUCH). The data is split by anatomical region 6('abdomen', 'thorax') and by 'train' / 'test' subset. 7 8This dataset is from the publication https://doi.org/10.1038/s41597-026-06597-6. 9Please cite it if you use this dataset in your research. 10""" 11 12import os 13from glob import glob 14from natsort import natsorted 15from typing import Union, Tuple, List, Literal, Optional 16 17from torch.utils.data import Dataset, DataLoader 18 19import torch_em 20 21from .. import util 22 23 24URL = "https://zenodo.org/records/17788162/files/final-formatted.zip" 25CHECKSUM = "39e2fc8a34ef7f519617901e4683462a75c4d97e5c46eedd50d3b35b22b96be5" 26 27REGIONS = ["abdomen", "thorax"] 28SPLITS = ["train", "test"] 29 30 31def get_recist_ct_data(path: Union[os.PathLike, str], download: bool = False) -> str: 32 """Download the RECIST CT dataset. 33 34 Args: 35 path: Filepath to a folder where the data is downloaded for further processing. 36 download: Whether to download the data if it is not present. 37 38 Returns: 39 Filepath where the data is stored. 40 """ 41 data_dir = os.path.join(path, "final-formatted") 42 if os.path.exists(data_dir): 43 return data_dir 44 45 os.makedirs(path, exist_ok=True) 46 47 zip_path = os.path.join(path, "final-formatted.zip") 48 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 49 util.unzip(zip_path=zip_path, dst=path) 50 51 return data_dir 52 53 54def get_recist_ct_paths( 55 path: Union[os.PathLike, str], 56 region: Optional[Literal["abdomen", "thorax"]] = None, 57 split: Optional[Literal["train", "test"]] = None, 58 download: bool = False, 59) -> Tuple[List[str], List[str]]: 60 """Get paths to the RECIST CT data. 61 62 Args: 63 path: Filepath to a folder where the data is downloaded for further processing. 64 region: The choice of anatomical region. 65 split: The choice of data split. 66 download: Whether to download the data if it is not present. 67 68 Returns: 69 List of filepaths for the image data. 70 List of filepaths for the label data. 71 """ 72 data_dir = get_recist_ct_data(path, download) 73 74 if region is None: 75 regions = REGIONS 76 else: 77 assert region in REGIONS, f"'{region}' is not a valid region." 78 regions = [region] 79 80 if split is None: 81 splits = SPLITS 82 else: 83 assert split in SPLITS, f"'{split}' is not a valid split." 84 splits = [split] 85 86 # The original filenames combine the patient id with the DICOM Series Instance UID, which contains many 87 # dots. 'elf.io.open_file' only recognizes '.nii.gz' files that have exactly two suffixes, so symlinks 88 # with the extra dots replaced by underscores are created here to make the files resolvable. 89 clean_dir = os.path.join(path, "clean") 90 91 image_paths, gt_paths = [], [] 92 for _region in regions: 93 for _split in splits: 94 base_dir = os.path.join(data_dir, "images", _region, _split) 95 raw_image_paths = natsorted(glob(os.path.join(base_dir, "images", "*.nii.gz"))) 96 raw_gt_paths = natsorted(glob(os.path.join(base_dir, "masks", "*.nii.gz"))) 97 98 for raw_path, sub_dir, out_paths in [ 99 (raw_image_paths, "images", image_paths), (raw_gt_paths, "masks", gt_paths) 100 ]: 101 out_dir = os.path.join(clean_dir, _region, _split, sub_dir) 102 os.makedirs(out_dir, exist_ok=True) 103 for orig_path in raw_path: 104 clean_name = os.path.basename(orig_path)[:-len(".nii.gz")].replace(".", "_") + ".nii.gz" 105 clean_path = os.path.join(out_dir, clean_name) 106 if not os.path.exists(clean_path): 107 os.symlink(os.path.abspath(orig_path), clean_path) 108 out_paths.append(clean_path) 109 110 assert len(image_paths) > 0 and len(image_paths) == len(gt_paths), \ 111 f"Could not find a matching number of images and labels in '{data_dir}'." 112 113 return image_paths, gt_paths 114 115 116def get_recist_ct_dataset( 117 path: Union[os.PathLike, str], 118 patch_shape: Tuple[int, ...], 119 region: Optional[Literal["abdomen", "thorax"]] = None, 120 split: Optional[Literal["train", "test"]] = None, 121 resize_inputs: bool = False, 122 download: bool = False, 123 **kwargs 124) -> Dataset: 125 """Get the RECIST CT dataset for instance segmentation of tumors, metastases and lymph nodes. 126 127 Args: 128 path: Filepath to a folder where the data is downloaded for further processing. 129 patch_shape: The patch shape to use for training. 130 region: The choice of anatomical region. 131 split: The choice of data split. 132 resize_inputs: Whether to resize the inputs to the patch shape. 133 download: Whether to download the data if it is not present. 134 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 135 136 Returns: 137 The segmentation dataset. 138 """ 139 image_paths, gt_paths = get_recist_ct_paths(path, region, split, download) 140 141 if resize_inputs: 142 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 143 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 144 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 145 ) 146 147 return torch_em.default_segmentation_dataset( 148 raw_paths=image_paths, 149 raw_key="data", 150 label_paths=gt_paths, 151 label_key="data", 152 patch_shape=patch_shape, 153 is_seg_dataset=True, 154 **kwargs 155 ) 156 157 158def get_recist_ct_loader( 159 path: Union[os.PathLike, str], 160 batch_size: int, 161 patch_shape: Tuple[int, ...], 162 region: Optional[Literal["abdomen", "thorax"]] = None, 163 split: Optional[Literal["train", "test"]] = None, 164 resize_inputs: bool = False, 165 download: bool = False, 166 **kwargs 167) -> DataLoader: 168 """Get the RECIST CT dataloader for instance segmentation of tumors, metastases and lymph nodes. 169 170 Args: 171 path: Filepath to a folder where the data is downloaded for further processing. 172 batch_size: The batch size for training. 173 patch_shape: The patch shape to use for training. 174 region: The choice of anatomical region. 175 split: The choice of data split. 176 resize_inputs: Whether to resize the inputs to the patch shape. 177 download: Whether to download the data if it is not present. 178 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or the PyTorch DataLoader. 179 180 Returns: 181 The DataLoader. 182 """ 183 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 184 dataset = get_recist_ct_dataset(path, patch_shape, region, split, resize_inputs, download, **ds_kwargs) 185 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
32def get_recist_ct_data(path: Union[os.PathLike, str], download: bool = False) -> str: 33 """Download the RECIST CT dataset. 34 35 Args: 36 path: Filepath to a folder where the data is downloaded for further processing. 37 download: Whether to download the data if it is not present. 38 39 Returns: 40 Filepath where the data is stored. 41 """ 42 data_dir = os.path.join(path, "final-formatted") 43 if os.path.exists(data_dir): 44 return data_dir 45 46 os.makedirs(path, exist_ok=True) 47 48 zip_path = os.path.join(path, "final-formatted.zip") 49 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 50 util.unzip(zip_path=zip_path, dst=path) 51 52 return data_dir
Download the RECIST CT dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is stored.
55def get_recist_ct_paths( 56 path: Union[os.PathLike, str], 57 region: Optional[Literal["abdomen", "thorax"]] = None, 58 split: Optional[Literal["train", "test"]] = None, 59 download: bool = False, 60) -> Tuple[List[str], List[str]]: 61 """Get paths to the RECIST CT data. 62 63 Args: 64 path: Filepath to a folder where the data is downloaded for further processing. 65 region: The choice of anatomical region. 66 split: The choice of data split. 67 download: Whether to download the data if it is not present. 68 69 Returns: 70 List of filepaths for the image data. 71 List of filepaths for the label data. 72 """ 73 data_dir = get_recist_ct_data(path, download) 74 75 if region is None: 76 regions = REGIONS 77 else: 78 assert region in REGIONS, f"'{region}' is not a valid region." 79 regions = [region] 80 81 if split is None: 82 splits = SPLITS 83 else: 84 assert split in SPLITS, f"'{split}' is not a valid split." 85 splits = [split] 86 87 # The original filenames combine the patient id with the DICOM Series Instance UID, which contains many 88 # dots. 'elf.io.open_file' only recognizes '.nii.gz' files that have exactly two suffixes, so symlinks 89 # with the extra dots replaced by underscores are created here to make the files resolvable. 90 clean_dir = os.path.join(path, "clean") 91 92 image_paths, gt_paths = [], [] 93 for _region in regions: 94 for _split in splits: 95 base_dir = os.path.join(data_dir, "images", _region, _split) 96 raw_image_paths = natsorted(glob(os.path.join(base_dir, "images", "*.nii.gz"))) 97 raw_gt_paths = natsorted(glob(os.path.join(base_dir, "masks", "*.nii.gz"))) 98 99 for raw_path, sub_dir, out_paths in [ 100 (raw_image_paths, "images", image_paths), (raw_gt_paths, "masks", gt_paths) 101 ]: 102 out_dir = os.path.join(clean_dir, _region, _split, sub_dir) 103 os.makedirs(out_dir, exist_ok=True) 104 for orig_path in raw_path: 105 clean_name = os.path.basename(orig_path)[:-len(".nii.gz")].replace(".", "_") + ".nii.gz" 106 clean_path = os.path.join(out_dir, clean_name) 107 if not os.path.exists(clean_path): 108 os.symlink(os.path.abspath(orig_path), clean_path) 109 out_paths.append(clean_path) 110 111 assert len(image_paths) > 0 and len(image_paths) == len(gt_paths), \ 112 f"Could not find a matching number of images and labels in '{data_dir}'." 113 114 return image_paths, gt_paths
Get paths to the RECIST CT data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- region: The choice of anatomical region.
- split: The choice of data split.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
117def get_recist_ct_dataset( 118 path: Union[os.PathLike, str], 119 patch_shape: Tuple[int, ...], 120 region: Optional[Literal["abdomen", "thorax"]] = None, 121 split: Optional[Literal["train", "test"]] = None, 122 resize_inputs: bool = False, 123 download: bool = False, 124 **kwargs 125) -> Dataset: 126 """Get the RECIST CT dataset for instance segmentation of tumors, metastases and lymph nodes. 127 128 Args: 129 path: Filepath to a folder where the data is downloaded for further processing. 130 patch_shape: The patch shape to use for training. 131 region: The choice of anatomical region. 132 split: The choice of data split. 133 resize_inputs: Whether to resize the inputs to the patch shape. 134 download: Whether to download the data if it is not present. 135 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 136 137 Returns: 138 The segmentation dataset. 139 """ 140 image_paths, gt_paths = get_recist_ct_paths(path, region, split, download) 141 142 if resize_inputs: 143 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 144 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 145 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 146 ) 147 148 return torch_em.default_segmentation_dataset( 149 raw_paths=image_paths, 150 raw_key="data", 151 label_paths=gt_paths, 152 label_key="data", 153 patch_shape=patch_shape, 154 is_seg_dataset=True, 155 **kwargs 156 )
Get the RECIST CT dataset for instance segmentation of tumors, metastases and lymph nodes.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- region: The choice of anatomical region.
- split: The choice of data split.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
159def get_recist_ct_loader( 160 path: Union[os.PathLike, str], 161 batch_size: int, 162 patch_shape: Tuple[int, ...], 163 region: Optional[Literal["abdomen", "thorax"]] = None, 164 split: Optional[Literal["train", "test"]] = None, 165 resize_inputs: bool = False, 166 download: bool = False, 167 **kwargs 168) -> DataLoader: 169 """Get the RECIST CT dataloader for instance segmentation of tumors, metastases and lymph nodes. 170 171 Args: 172 path: Filepath to a folder where the data is downloaded for further processing. 173 batch_size: The batch size for training. 174 patch_shape: The patch shape to use for training. 175 region: The choice of anatomical region. 176 split: The choice of data split. 177 resize_inputs: Whether to resize the inputs to the patch shape. 178 download: Whether to download the data if it is not present. 179 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or the PyTorch DataLoader. 180 181 Returns: 182 The DataLoader. 183 """ 184 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 185 dataset = get_recist_ct_dataset(path, patch_shape, region, split, resize_inputs, download, **ds_kwargs) 186 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the RECIST CT dataloader for instance segmentation of tumors, metastases and lymph nodes.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- region: The choice of anatomical region.
- split: The choice of data split.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor the PyTorch DataLoader.
Returns:
The DataLoader.