torch_em.data.datasets.medical.trusted

The TRUSTED dataset contains annotations for kidney segmentation in paired 3d transabdominal ultrasound and CT volumes.

The dataset contains 96 kidneys from 48 human patients, imaged with both transabdominal 3d ultrasound and CT. Two independent radiographers manually segmented each kidney, and a STAPLE-fused consensus segmentation is provided in addition to the individual annotations.

This dataset is located at https://doi.org/10.6084/m9.figshare.27981050.v1 (CC BY 4.0). This dataset is from the publication https://doi.org/10.1038/s41597-025-04467-1. Please cite it if you use this dataset for your research.

  1"""The TRUSTED dataset contains annotations for kidney segmentation in paired 3d
  2transabdominal ultrasound and CT volumes.
  3
  4The dataset contains 96 kidneys from 48 human patients, imaged with both transabdominal
  53d ultrasound and CT. Two independent radiographers manually segmented each kidney, and
  6a STAPLE-fused consensus segmentation is provided in addition to the individual annotations.
  7
  8This dataset is located at https://doi.org/10.6084/m9.figshare.27981050.v1 (CC BY 4.0).
  9This dataset is from the publication https://doi.org/10.1038/s41597-025-04467-1.
 10Please cite it if you use this dataset for your research.
 11"""
 12
 13import os
 14from glob import glob
 15from typing import Union, Tuple, Literal, List
 16
 17from torch.utils.data import Dataset, DataLoader
 18
 19import torch_em
 20
 21from .. import util
 22
 23
 24URL = "https://ndownloader.figshare.com/files/51079133"
 25CHECKSUM = "2e63e560f4dcbccba920cd90e2add9738da01efa8e5f6834bd36c4768757b0b4"
 26
 27# The estimated STAPLE-fused consensus masks contain tiny near-zero floating point noise
 28# (e.g. ~1e-14) alongside actual foreground values (~1.0) instead of clean binary values.
 29# We binarize the labels on the fly (per sampled patch) rather than rewriting the full
 30# 3d volumes to disk, since the volumes are large (up to ~300MB each, uncompressed GBs).
 31
 32
 33def _binarize_labels(labels):
 34    return (labels > 0.5).astype("float32")
 35
 36
 37def get_trusted_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 38    """Download the TRUSTED dataset.
 39
 40    Args:
 41        path: Filepath to a folder where the data is downloaded for further processing.
 42        download: Whether to download the data if it is not present.
 43
 44    Returns:
 45        Filepath where the data is downloaded.
 46    """
 47    data_dir = os.path.join(path, "TRUSTED_dataset_for_nsd")
 48    if os.path.exists(data_dir):
 49        return data_dir
 50
 51    os.makedirs(path, exist_ok=True)
 52
 53    zip_path = os.path.join(path, "TRUSTED_dataset_for_nsd.zip")
 54    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 55    util.unzip(zip_path=zip_path, dst=path)
 56
 57    return data_dir
 58
 59
 60def get_trusted_paths(
 61    path: Union[os.PathLike, str],
 62    modality: Literal["us", "ct"] = "us",
 63    label_choice: Literal["gt_estimated", "annotator1", "annotator2"] = "gt_estimated",
 64    download: bool = False,
 65) -> Tuple[List[str], List[str]]:
 66    """Get paths to the TRUSTED data.
 67
 68    Args:
 69        path: Filepath to a folder where the data is downloaded for further processing.
 70        modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct').
 71        label_choice: The choice of segmentation annotation to use, either the STAPLE-fused
 72            consensus mask ('gt_estimated') or one of the two individual annotators
 73            ('annotator1' / 'annotator2').
 74        download: Whether to download the data if it is not present.
 75
 76    Returns:
 77        List of filepaths for the image data.
 78        List of filepaths for the label data.
 79    """
 80    data_dir = get_trusted_data(path=path, download=download)
 81
 82    if modality not in ["us", "ct"]:
 83        raise ValueError(f"'{modality}' is not a valid modality choice.")
 84
 85    if label_choice not in ["gt_estimated", "annotator1", "annotator2"]:
 86        raise ValueError(f"'{label_choice}' is not a valid label choice.")
 87
 88    suffix = modality.upper()
 89    image_dir = os.path.join(data_dir, f"{suffix}_DATA", f"{suffix}_images")
 90    mask_dir = os.path.join(data_dir, f"{suffix}_DATA", f"{suffix}_masks")
 91
 92    all_image_paths = sorted(glob(os.path.join(image_dir, f"*_img{suffix}.nii.gz")))
 93
 94    image_paths, gt_paths = [], []
 95    for image_path in all_image_paths:
 96        case_id = os.path.basename(image_path).replace(f"_img{suffix}.nii.gz", "")
 97        if label_choice == "gt_estimated":
 98            gt_path = os.path.join(mask_dir, f"GT_estimated_masks{suffix}", f"{case_id}_mask{suffix}.nii.gz")
 99        elif label_choice == "annotator1":
100            annotator_id = case_id + "1" if modality == "us" else case_id + "_1"
101            gt_path = os.path.join(mask_dir, "Annotator1", f"{annotator_id}_mask{suffix}.nii.gz")
102        else:
103            annotator_id = case_id + "2" if modality == "us" else case_id + "_2"
104            gt_path = os.path.join(mask_dir, "Annotator2", f"{annotator_id}_mask{suffix}.nii.gz")
105
106        # Not every case has annotations from both annotators, so we skip the ones that are missing.
107        if os.path.exists(gt_path):
108            image_paths.append(image_path)
109            gt_paths.append(gt_path)
110
111    if len(image_paths) == 0 or len(image_paths) != len(gt_paths):
112        raise RuntimeError("Something went wrong with fetching the image and label paths.")
113
114    return image_paths, gt_paths
115
116
117def get_trusted_dataset(
118    path: Union[os.PathLike, str],
119    patch_shape: Tuple[int, int, int],
120    modality: Literal["us", "ct"] = "us",
121    label_choice: Literal["gt_estimated", "annotator1", "annotator2"] = "gt_estimated",
122    resize_inputs: bool = False,
123    download: bool = False,
124    **kwargs
125) -> Dataset:
126    """Get the TRUSTED dataset for kidney segmentation.
127
128    Args:
129        path: Filepath to a folder where the data is downloaded for further processing.
130        patch_shape: The patch shape to use for training.
131        modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct').
132        label_choice: The choice of segmentation annotation to use, either the STAPLE-fused
133            consensus mask ('gt_estimated') or one of the two individual annotators
134            ('annotator1' / 'annotator2').
135        resize_inputs: Whether to resize the inputs.
136        download: Whether to download the data if it is not present.
137        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
138
139    Returns:
140        The segmentation dataset.
141    """
142    image_paths, gt_paths = get_trusted_paths(path, modality, label_choice, download)
143
144    if resize_inputs:
145        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
146        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
147            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
148        )
149
150    kwargs.setdefault("pre_label_transform", _binarize_labels)
151
152    return torch_em.default_segmentation_dataset(
153        raw_paths=image_paths,
154        raw_key="data",
155        label_paths=gt_paths,
156        label_key="data",
157        patch_shape=patch_shape,
158        **kwargs
159    )
160
161
162def get_trusted_loader(
163    path: Union[os.PathLike, str],
164    batch_size: int,
165    patch_shape: Tuple[int, int, int],
166    modality: Literal["us", "ct"] = "us",
167    label_choice: Literal["gt_estimated", "annotator1", "annotator2"] = "gt_estimated",
168    resize_inputs: bool = False,
169    download: bool = False,
170    **kwargs
171) -> DataLoader:
172    """Get the TRUSTED dataloader for kidney segmentation.
173
174    Args:
175        path: Filepath to a folder where the data is downloaded for further processing.
176        batch_size: The batch size for training.
177        patch_shape: The patch shape to use for training.
178        modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct').
179        label_choice: The choice of segmentation annotation to use, either the STAPLE-fused
180            consensus mask ('gt_estimated') or one of the two individual annotators
181            ('annotator1' / 'annotator2').
182        resize_inputs: Whether to resize the inputs.
183        download: Whether to download the data if it is not present.
184        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
185
186    Returns:
187        The DataLoader.
188    """
189    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
190    dataset = get_trusted_dataset(path, patch_shape, modality, label_choice, resize_inputs, download, **ds_kwargs)
191    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://ndownloader.figshare.com/files/51079133'
CHECKSUM = '2e63e560f4dcbccba920cd90e2add9738da01efa8e5f6834bd36c4768757b0b4'
def get_trusted_data(path: Union[os.PathLike, str], download: bool = False) -> str:
38def get_trusted_data(path: Union[os.PathLike, str], download: bool = False) -> str:
39    """Download the TRUSTED dataset.
40
41    Args:
42        path: Filepath to a folder where the data is downloaded for further processing.
43        download: Whether to download the data if it is not present.
44
45    Returns:
46        Filepath where the data is downloaded.
47    """
48    data_dir = os.path.join(path, "TRUSTED_dataset_for_nsd")
49    if os.path.exists(data_dir):
50        return data_dir
51
52    os.makedirs(path, exist_ok=True)
53
54    zip_path = os.path.join(path, "TRUSTED_dataset_for_nsd.zip")
55    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
56    util.unzip(zip_path=zip_path, dst=path)
57
58    return data_dir

Download the TRUSTED dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_trusted_paths( path: Union[os.PathLike, str], modality: Literal['us', 'ct'] = 'us', label_choice: Literal['gt_estimated', 'annotator1', 'annotator2'] = 'gt_estimated', download: bool = False) -> Tuple[List[str], List[str]]:
 61def get_trusted_paths(
 62    path: Union[os.PathLike, str],
 63    modality: Literal["us", "ct"] = "us",
 64    label_choice: Literal["gt_estimated", "annotator1", "annotator2"] = "gt_estimated",
 65    download: bool = False,
 66) -> Tuple[List[str], List[str]]:
 67    """Get paths to the TRUSTED data.
 68
 69    Args:
 70        path: Filepath to a folder where the data is downloaded for further processing.
 71        modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct').
 72        label_choice: The choice of segmentation annotation to use, either the STAPLE-fused
 73            consensus mask ('gt_estimated') or one of the two individual annotators
 74            ('annotator1' / 'annotator2').
 75        download: Whether to download the data if it is not present.
 76
 77    Returns:
 78        List of filepaths for the image data.
 79        List of filepaths for the label data.
 80    """
 81    data_dir = get_trusted_data(path=path, download=download)
 82
 83    if modality not in ["us", "ct"]:
 84        raise ValueError(f"'{modality}' is not a valid modality choice.")
 85
 86    if label_choice not in ["gt_estimated", "annotator1", "annotator2"]:
 87        raise ValueError(f"'{label_choice}' is not a valid label choice.")
 88
 89    suffix = modality.upper()
 90    image_dir = os.path.join(data_dir, f"{suffix}_DATA", f"{suffix}_images")
 91    mask_dir = os.path.join(data_dir, f"{suffix}_DATA", f"{suffix}_masks")
 92
 93    all_image_paths = sorted(glob(os.path.join(image_dir, f"*_img{suffix}.nii.gz")))
 94
 95    image_paths, gt_paths = [], []
 96    for image_path in all_image_paths:
 97        case_id = os.path.basename(image_path).replace(f"_img{suffix}.nii.gz", "")
 98        if label_choice == "gt_estimated":
 99            gt_path = os.path.join(mask_dir, f"GT_estimated_masks{suffix}", f"{case_id}_mask{suffix}.nii.gz")
100        elif label_choice == "annotator1":
101            annotator_id = case_id + "1" if modality == "us" else case_id + "_1"
102            gt_path = os.path.join(mask_dir, "Annotator1", f"{annotator_id}_mask{suffix}.nii.gz")
103        else:
104            annotator_id = case_id + "2" if modality == "us" else case_id + "_2"
105            gt_path = os.path.join(mask_dir, "Annotator2", f"{annotator_id}_mask{suffix}.nii.gz")
106
107        # Not every case has annotations from both annotators, so we skip the ones that are missing.
108        if os.path.exists(gt_path):
109            image_paths.append(image_path)
110            gt_paths.append(gt_path)
111
112    if len(image_paths) == 0 or len(image_paths) != len(gt_paths):
113        raise RuntimeError("Something went wrong with fetching the image and label paths.")
114
115    return image_paths, gt_paths

Get paths to the TRUSTED data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct').
  • label_choice: The choice of segmentation annotation to use, either the STAPLE-fused consensus mask ('gt_estimated') or one of the two individual annotators ('annotator1' / 'annotator2').
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_trusted_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], modality: Literal['us', 'ct'] = 'us', label_choice: Literal['gt_estimated', 'annotator1', 'annotator2'] = 'gt_estimated', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
118def get_trusted_dataset(
119    path: Union[os.PathLike, str],
120    patch_shape: Tuple[int, int, int],
121    modality: Literal["us", "ct"] = "us",
122    label_choice: Literal["gt_estimated", "annotator1", "annotator2"] = "gt_estimated",
123    resize_inputs: bool = False,
124    download: bool = False,
125    **kwargs
126) -> Dataset:
127    """Get the TRUSTED dataset for kidney segmentation.
128
129    Args:
130        path: Filepath to a folder where the data is downloaded for further processing.
131        patch_shape: The patch shape to use for training.
132        modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct').
133        label_choice: The choice of segmentation annotation to use, either the STAPLE-fused
134            consensus mask ('gt_estimated') or one of the two individual annotators
135            ('annotator1' / 'annotator2').
136        resize_inputs: Whether to resize the inputs.
137        download: Whether to download the data if it is not present.
138        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
139
140    Returns:
141        The segmentation dataset.
142    """
143    image_paths, gt_paths = get_trusted_paths(path, modality, label_choice, download)
144
145    if resize_inputs:
146        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
147        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
148            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
149        )
150
151    kwargs.setdefault("pre_label_transform", _binarize_labels)
152
153    return torch_em.default_segmentation_dataset(
154        raw_paths=image_paths,
155        raw_key="data",
156        label_paths=gt_paths,
157        label_key="data",
158        patch_shape=patch_shape,
159        **kwargs
160    )

Get the TRUSTED dataset for kidney segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct').
  • label_choice: The choice of segmentation annotation to use, either the STAPLE-fused consensus mask ('gt_estimated') or one of the two individual annotators ('annotator1' / 'annotator2').
  • resize_inputs: Whether to resize the inputs.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_trusted_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int, int], modality: Literal['us', 'ct'] = 'us', label_choice: Literal['gt_estimated', 'annotator1', 'annotator2'] = 'gt_estimated', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
163def get_trusted_loader(
164    path: Union[os.PathLike, str],
165    batch_size: int,
166    patch_shape: Tuple[int, int, int],
167    modality: Literal["us", "ct"] = "us",
168    label_choice: Literal["gt_estimated", "annotator1", "annotator2"] = "gt_estimated",
169    resize_inputs: bool = False,
170    download: bool = False,
171    **kwargs
172) -> DataLoader:
173    """Get the TRUSTED dataloader for kidney segmentation.
174
175    Args:
176        path: Filepath to a folder where the data is downloaded for further processing.
177        batch_size: The batch size for training.
178        patch_shape: The patch shape to use for training.
179        modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct').
180        label_choice: The choice of segmentation annotation to use, either the STAPLE-fused
181            consensus mask ('gt_estimated') or one of the two individual annotators
182            ('annotator1' / 'annotator2').
183        resize_inputs: Whether to resize the inputs.
184        download: Whether to download the data if it is not present.
185        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
186
187    Returns:
188        The DataLoader.
189    """
190    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
191    dataset = get_trusted_dataset(path, patch_shape, modality, label_choice, resize_inputs, download, **ds_kwargs)
192    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the TRUSTED dataloader for kidney segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct').
  • label_choice: The choice of segmentation annotation to use, either the STAPLE-fused consensus mask ('gt_estimated') or one of the two individual annotators ('annotator1' / 'annotator2').
  • resize_inputs: Whether to resize the inputs.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.