torch_em.data.datasets.medical.rcc_aid

The RCC-AID dataset contains annotations for kidney, tumor and cyst segmentation in CT scans of patients with renal cell carcinoma (RCC).

The dataset consists of 129 CT series from 91 patients across the TCGA-KIRC (clear cell, 85 series), TCGA-KIRP (papillary, 26 series) and TCGA-KICH (chromophobe, 18 series) cohorts. The images were converted from the original TCIA DICOM series to NIfTI, and voxel-level segmentation masks were derived from an automated segmentation model followed by manual quality checks and corrections.

The label ids are - kidney: 1, tumor: 2, cyst: 3

This dataset is from the publication https://doi.org/10.64898/2026.04.22.26351451. Please cite it if you use this dataset in your research.

  1"""The RCC-AID dataset contains annotations for kidney, tumor and cyst segmentation in CT scans of
  2patients with renal cell carcinoma (RCC).
  3
  4The dataset consists of 129 CT series from 91 patients across the TCGA-KIRC (clear cell, 85 series),
  5TCGA-KIRP (papillary, 26 series) and TCGA-KICH (chromophobe, 18 series) cohorts. The images were
  6converted from the original TCIA DICOM series to NIfTI, and voxel-level segmentation masks were
  7derived from an automated segmentation model followed by manual quality checks and corrections.
  8
  9The label ids are - kidney: 1, tumor: 2, cyst: 3
 10
 11This dataset is from the publication https://doi.org/10.64898/2026.04.22.26351451.
 12Please cite it if you use this dataset in your research.
 13"""
 14
 15import os
 16from glob import glob
 17from natsort import natsorted
 18from typing import Union, Tuple, List
 19
 20from torch.utils.data import Dataset, DataLoader
 21
 22import torch_em
 23
 24from .. import util
 25
 26
 27URL = {
 28    "images": "https://zenodo.org/records/20719257/files/images.zip",
 29    "labels": "https://zenodo.org/records/20719257/files/labels.zip",
 30}
 31
 32CHECKSUM = {
 33    "images": "7e6a7530e8bce0c34df1cafe00a56bbe579f480b54a9e963d206fe385ff23ade",
 34    "labels": "692cae50e2de498b63fc6edcc526a57ae19fecfbaadd6f6ad0a4ff928c2da6d1",
 35}
 36
 37
 38def get_rcc_aid_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 39    """Download the RCC-AID dataset.
 40
 41    Args:
 42        path: Filepath to a folder where the data is downloaded for further processing.
 43        download: Whether to download the data if it is not present.
 44
 45    Returns:
 46        Filepath where the data is stored.
 47    """
 48    images_dir = os.path.join(path, "images")
 49    labels_dir = os.path.join(path, "labels")
 50    if os.path.exists(images_dir) and os.path.exists(labels_dir):
 51        return path
 52
 53    os.makedirs(path, exist_ok=True)
 54
 55    for name, out_dir in [("images", images_dir), ("labels", labels_dir)]:
 56        zip_path = os.path.join(path, f"{name}.zip")
 57        util.download_source(path=zip_path, url=URL[name], download=download, checksum=CHECKSUM[name])
 58        util.unzip(zip_path=zip_path, dst=path)
 59        assert os.path.exists(out_dir), f"Something went wrong when unzipping '{zip_path}'."
 60
 61    return path
 62
 63
 64def get_rcc_aid_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 65    """Get paths to the RCC-AID data.
 66
 67    Args:
 68        path: Filepath to a folder where the data is downloaded for further processing.
 69        download: Whether to download the data if it is not present.
 70
 71    Returns:
 72        List of filepaths for the image data.
 73        List of filepaths for the label data.
 74    """
 75    data_dir = get_rcc_aid_data(path, download)
 76
 77    label_paths = natsorted(glob(os.path.join(data_dir, "labels", "*_seg.nii.gz")))
 78    image_paths = [
 79        os.path.join(data_dir, "images", os.path.basename(p)[:-len("_seg.nii.gz")] + ".nii.gz")
 80        for p in label_paths
 81    ]
 82
 83    missing = [p for p in image_paths if not os.path.exists(p)]
 84    assert not missing, f"Could not find the matching image(s) for {missing}."
 85
 86    return image_paths, label_paths
 87
 88
 89def get_rcc_aid_dataset(
 90    path: Union[os.PathLike, str],
 91    patch_shape: Tuple[int, ...],
 92    resize_inputs: bool = False,
 93    download: bool = False,
 94    **kwargs
 95) -> Dataset:
 96    """Get the RCC-AID dataset for kidney, tumor and cyst segmentation.
 97
 98    Args:
 99        path: Filepath to a folder where the data is downloaded for further processing.
100        patch_shape: The patch shape to use for training.
101        resize_inputs: Whether to resize the inputs to the patch shape.
102        download: Whether to download the data if it is not present.
103        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
104
105    Returns:
106        The segmentation dataset.
107    """
108    image_paths, label_paths = get_rcc_aid_paths(path, download)
109
110    if resize_inputs:
111        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
112        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
113            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
114        )
115
116    return torch_em.default_segmentation_dataset(
117        raw_paths=image_paths,
118        raw_key="data",
119        label_paths=label_paths,
120        label_key="data",
121        patch_shape=patch_shape,
122        is_seg_dataset=True,
123        **kwargs
124    )
125
126
127def get_rcc_aid_loader(
128    path: Union[os.PathLike, str],
129    batch_size: int,
130    patch_shape: Tuple[int, ...],
131    resize_inputs: bool = False,
132    download: bool = False,
133    **kwargs
134) -> DataLoader:
135    """Get the RCC-AID dataloader for kidney, tumor and cyst segmentation.
136
137    Args:
138        path: Filepath to a folder where the data is downloaded for further processing.
139        batch_size: The batch size for training.
140        patch_shape: The patch shape to use for training.
141        resize_inputs: Whether to resize the inputs to the patch shape.
142        download: Whether to download the data if it is not present.
143        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or the PyTorch DataLoader.
144
145    Returns:
146        The DataLoader.
147    """
148    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
149    dataset = get_rcc_aid_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
150    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = {'images': 'https://zenodo.org/records/20719257/files/images.zip', 'labels': 'https://zenodo.org/records/20719257/files/labels.zip'}
CHECKSUM = {'images': '7e6a7530e8bce0c34df1cafe00a56bbe579f480b54a9e963d206fe385ff23ade', 'labels': '692cae50e2de498b63fc6edcc526a57ae19fecfbaadd6f6ad0a4ff928c2da6d1'}
def get_rcc_aid_data(path: Union[os.PathLike, str], download: bool = False) -> str:
39def get_rcc_aid_data(path: Union[os.PathLike, str], download: bool = False) -> str:
40    """Download the RCC-AID dataset.
41
42    Args:
43        path: Filepath to a folder where the data is downloaded for further processing.
44        download: Whether to download the data if it is not present.
45
46    Returns:
47        Filepath where the data is stored.
48    """
49    images_dir = os.path.join(path, "images")
50    labels_dir = os.path.join(path, "labels")
51    if os.path.exists(images_dir) and os.path.exists(labels_dir):
52        return path
53
54    os.makedirs(path, exist_ok=True)
55
56    for name, out_dir in [("images", images_dir), ("labels", labels_dir)]:
57        zip_path = os.path.join(path, f"{name}.zip")
58        util.download_source(path=zip_path, url=URL[name], download=download, checksum=CHECKSUM[name])
59        util.unzip(zip_path=zip_path, dst=path)
60        assert os.path.exists(out_dir), f"Something went wrong when unzipping '{zip_path}'."
61
62    return path

Download the RCC-AID dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is stored.

def get_rcc_aid_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
65def get_rcc_aid_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
66    """Get paths to the RCC-AID data.
67
68    Args:
69        path: Filepath to a folder where the data is downloaded for further processing.
70        download: Whether to download the data if it is not present.
71
72    Returns:
73        List of filepaths for the image data.
74        List of filepaths for the label data.
75    """
76    data_dir = get_rcc_aid_data(path, download)
77
78    label_paths = natsorted(glob(os.path.join(data_dir, "labels", "*_seg.nii.gz")))
79    image_paths = [
80        os.path.join(data_dir, "images", os.path.basename(p)[:-len("_seg.nii.gz")] + ".nii.gz")
81        for p in label_paths
82    ]
83
84    missing = [p for p in image_paths if not os.path.exists(p)]
85    assert not missing, f"Could not find the matching image(s) for {missing}."
86
87    return image_paths, label_paths

Get paths to the RCC-AID data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_rcc_aid_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
 90def get_rcc_aid_dataset(
 91    path: Union[os.PathLike, str],
 92    patch_shape: Tuple[int, ...],
 93    resize_inputs: bool = False,
 94    download: bool = False,
 95    **kwargs
 96) -> Dataset:
 97    """Get the RCC-AID dataset for kidney, tumor and cyst segmentation.
 98
 99    Args:
100        path: Filepath to a folder where the data is downloaded for further processing.
101        patch_shape: The patch shape to use for training.
102        resize_inputs: Whether to resize the inputs to the patch shape.
103        download: Whether to download the data if it is not present.
104        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
105
106    Returns:
107        The segmentation dataset.
108    """
109    image_paths, label_paths = get_rcc_aid_paths(path, download)
110
111    if resize_inputs:
112        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
113        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
114            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
115        )
116
117    return torch_em.default_segmentation_dataset(
118        raw_paths=image_paths,
119        raw_key="data",
120        label_paths=label_paths,
121        label_key="data",
122        patch_shape=patch_shape,
123        is_seg_dataset=True,
124        **kwargs
125    )

Get the RCC-AID dataset for kidney, tumor and cyst segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_rcc_aid_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
128def get_rcc_aid_loader(
129    path: Union[os.PathLike, str],
130    batch_size: int,
131    patch_shape: Tuple[int, ...],
132    resize_inputs: bool = False,
133    download: bool = False,
134    **kwargs
135) -> DataLoader:
136    """Get the RCC-AID dataloader for kidney, tumor and cyst segmentation.
137
138    Args:
139        path: Filepath to a folder where the data is downloaded for further processing.
140        batch_size: The batch size for training.
141        patch_shape: The patch shape to use for training.
142        resize_inputs: Whether to resize the inputs to the patch shape.
143        download: Whether to download the data if it is not present.
144        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or the PyTorch DataLoader.
145
146    Returns:
147        The DataLoader.
148    """
149    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
150    dataset = get_rcc_aid_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
151    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the RCC-AID dataloader for kidney, tumor and cyst segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or the PyTorch DataLoader.
Returns:

The DataLoader.