torch_em.data.datasets.medical.trusted
The TRUSTED dataset contains annotations for kidney segmentation in paired 3d transabdominal ultrasound and CT volumes.
The dataset contains 96 kidneys from 48 human patients, imaged with both transabdominal 3d ultrasound and CT. Two independent radiographers manually segmented each kidney, and a STAPLE-fused consensus segmentation is provided in addition to the individual annotations.
This dataset is located at https://doi.org/10.6084/m9.figshare.27981050.v1 (CC BY 4.0). This dataset is from the publication https://doi.org/10.1038/s41597-025-04467-1. Please cite it if you use this dataset for your research.
1"""The TRUSTED dataset contains annotations for kidney segmentation in paired 3d 2transabdominal ultrasound and CT volumes. 3 4The dataset contains 96 kidneys from 48 human patients, imaged with both transabdominal 53d ultrasound and CT. Two independent radiographers manually segmented each kidney, and 6a STAPLE-fused consensus segmentation is provided in addition to the individual annotations. 7 8This dataset is located at https://doi.org/10.6084/m9.figshare.27981050.v1 (CC BY 4.0). 9This dataset is from the publication https://doi.org/10.1038/s41597-025-04467-1. 10Please cite it if you use this dataset for your research. 11""" 12 13import os 14from glob import glob 15from typing import Union, Tuple, Literal, List 16 17from torch.utils.data import Dataset, DataLoader 18 19import torch_em 20 21from .. import util 22 23 24URL = "https://ndownloader.figshare.com/files/51079133" 25CHECKSUM = "2e63e560f4dcbccba920cd90e2add9738da01efa8e5f6834bd36c4768757b0b4" 26 27# The estimated STAPLE-fused consensus masks contain tiny near-zero floating point noise 28# (e.g. ~1e-14) alongside actual foreground values (~1.0) instead of clean binary values. 29# We binarize the labels on the fly (per sampled patch) rather than rewriting the full 30# 3d volumes to disk, since the volumes are large (up to ~300MB each, uncompressed GBs). 31 32 33def _binarize_labels(labels): 34 return (labels > 0.5).astype("float32") 35 36 37def get_trusted_data(path: Union[os.PathLike, str], download: bool = False) -> str: 38 """Download the TRUSTED dataset. 39 40 Args: 41 path: Filepath to a folder where the data is downloaded for further processing. 42 download: Whether to download the data if it is not present. 43 44 Returns: 45 Filepath where the data is downloaded. 46 """ 47 data_dir = os.path.join(path, "TRUSTED_dataset_for_nsd") 48 if os.path.exists(data_dir): 49 return data_dir 50 51 os.makedirs(path, exist_ok=True) 52 53 zip_path = os.path.join(path, "TRUSTED_dataset_for_nsd.zip") 54 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 55 util.unzip(zip_path=zip_path, dst=path) 56 57 return data_dir 58 59 60def get_trusted_paths( 61 path: Union[os.PathLike, str], 62 modality: Literal["us", "ct"] = "us", 63 label_choice: Literal["gt_estimated", "annotator1", "annotator2"] = "gt_estimated", 64 download: bool = False, 65) -> Tuple[List[str], List[str]]: 66 """Get paths to the TRUSTED data. 67 68 Args: 69 path: Filepath to a folder where the data is downloaded for further processing. 70 modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct'). 71 label_choice: The choice of segmentation annotation to use, either the STAPLE-fused 72 consensus mask ('gt_estimated') or one of the two individual annotators 73 ('annotator1' / 'annotator2'). 74 download: Whether to download the data if it is not present. 75 76 Returns: 77 List of filepaths for the image data. 78 List of filepaths for the label data. 79 """ 80 data_dir = get_trusted_data(path=path, download=download) 81 82 if modality not in ["us", "ct"]: 83 raise ValueError(f"'{modality}' is not a valid modality choice.") 84 85 if label_choice not in ["gt_estimated", "annotator1", "annotator2"]: 86 raise ValueError(f"'{label_choice}' is not a valid label choice.") 87 88 suffix = modality.upper() 89 image_dir = os.path.join(data_dir, f"{suffix}_DATA", f"{suffix}_images") 90 mask_dir = os.path.join(data_dir, f"{suffix}_DATA", f"{suffix}_masks") 91 92 all_image_paths = sorted(glob(os.path.join(image_dir, f"*_img{suffix}.nii.gz"))) 93 94 image_paths, gt_paths = [], [] 95 for image_path in all_image_paths: 96 case_id = os.path.basename(image_path).replace(f"_img{suffix}.nii.gz", "") 97 if label_choice == "gt_estimated": 98 gt_path = os.path.join(mask_dir, f"GT_estimated_masks{suffix}", f"{case_id}_mask{suffix}.nii.gz") 99 elif label_choice == "annotator1": 100 annotator_id = case_id + "1" if modality == "us" else case_id + "_1" 101 gt_path = os.path.join(mask_dir, "Annotator1", f"{annotator_id}_mask{suffix}.nii.gz") 102 else: 103 annotator_id = case_id + "2" if modality == "us" else case_id + "_2" 104 gt_path = os.path.join(mask_dir, "Annotator2", f"{annotator_id}_mask{suffix}.nii.gz") 105 106 # Not every case has annotations from both annotators, so we skip the ones that are missing. 107 if os.path.exists(gt_path): 108 image_paths.append(image_path) 109 gt_paths.append(gt_path) 110 111 if len(image_paths) == 0 or len(image_paths) != len(gt_paths): 112 raise RuntimeError("Something went wrong with fetching the image and label paths.") 113 114 return image_paths, gt_paths 115 116 117def get_trusted_dataset( 118 path: Union[os.PathLike, str], 119 patch_shape: Tuple[int, int, int], 120 modality: Literal["us", "ct"] = "us", 121 label_choice: Literal["gt_estimated", "annotator1", "annotator2"] = "gt_estimated", 122 resize_inputs: bool = False, 123 download: bool = False, 124 **kwargs 125) -> Dataset: 126 """Get the TRUSTED dataset for kidney segmentation. 127 128 Args: 129 path: Filepath to a folder where the data is downloaded for further processing. 130 patch_shape: The patch shape to use for training. 131 modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct'). 132 label_choice: The choice of segmentation annotation to use, either the STAPLE-fused 133 consensus mask ('gt_estimated') or one of the two individual annotators 134 ('annotator1' / 'annotator2'). 135 resize_inputs: Whether to resize the inputs. 136 download: Whether to download the data if it is not present. 137 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 138 139 Returns: 140 The segmentation dataset. 141 """ 142 image_paths, gt_paths = get_trusted_paths(path, modality, label_choice, download) 143 144 if resize_inputs: 145 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 146 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 147 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 148 ) 149 150 kwargs.setdefault("pre_label_transform", _binarize_labels) 151 152 return torch_em.default_segmentation_dataset( 153 raw_paths=image_paths, 154 raw_key="data", 155 label_paths=gt_paths, 156 label_key="data", 157 patch_shape=patch_shape, 158 **kwargs 159 ) 160 161 162def get_trusted_loader( 163 path: Union[os.PathLike, str], 164 batch_size: int, 165 patch_shape: Tuple[int, int, int], 166 modality: Literal["us", "ct"] = "us", 167 label_choice: Literal["gt_estimated", "annotator1", "annotator2"] = "gt_estimated", 168 resize_inputs: bool = False, 169 download: bool = False, 170 **kwargs 171) -> DataLoader: 172 """Get the TRUSTED dataloader for kidney segmentation. 173 174 Args: 175 path: Filepath to a folder where the data is downloaded for further processing. 176 batch_size: The batch size for training. 177 patch_shape: The patch shape to use for training. 178 modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct'). 179 label_choice: The choice of segmentation annotation to use, either the STAPLE-fused 180 consensus mask ('gt_estimated') or one of the two individual annotators 181 ('annotator1' / 'annotator2'). 182 resize_inputs: Whether to resize the inputs. 183 download: Whether to download the data if it is not present. 184 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 185 186 Returns: 187 The DataLoader. 188 """ 189 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 190 dataset = get_trusted_dataset(path, patch_shape, modality, label_choice, resize_inputs, download, **ds_kwargs) 191 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
38def get_trusted_data(path: Union[os.PathLike, str], download: bool = False) -> str: 39 """Download the TRUSTED dataset. 40 41 Args: 42 path: Filepath to a folder where the data is downloaded for further processing. 43 download: Whether to download the data if it is not present. 44 45 Returns: 46 Filepath where the data is downloaded. 47 """ 48 data_dir = os.path.join(path, "TRUSTED_dataset_for_nsd") 49 if os.path.exists(data_dir): 50 return data_dir 51 52 os.makedirs(path, exist_ok=True) 53 54 zip_path = os.path.join(path, "TRUSTED_dataset_for_nsd.zip") 55 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 56 util.unzip(zip_path=zip_path, dst=path) 57 58 return data_dir
Download the TRUSTED dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
61def get_trusted_paths( 62 path: Union[os.PathLike, str], 63 modality: Literal["us", "ct"] = "us", 64 label_choice: Literal["gt_estimated", "annotator1", "annotator2"] = "gt_estimated", 65 download: bool = False, 66) -> Tuple[List[str], List[str]]: 67 """Get paths to the TRUSTED data. 68 69 Args: 70 path: Filepath to a folder where the data is downloaded for further processing. 71 modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct'). 72 label_choice: The choice of segmentation annotation to use, either the STAPLE-fused 73 consensus mask ('gt_estimated') or one of the two individual annotators 74 ('annotator1' / 'annotator2'). 75 download: Whether to download the data if it is not present. 76 77 Returns: 78 List of filepaths for the image data. 79 List of filepaths for the label data. 80 """ 81 data_dir = get_trusted_data(path=path, download=download) 82 83 if modality not in ["us", "ct"]: 84 raise ValueError(f"'{modality}' is not a valid modality choice.") 85 86 if label_choice not in ["gt_estimated", "annotator1", "annotator2"]: 87 raise ValueError(f"'{label_choice}' is not a valid label choice.") 88 89 suffix = modality.upper() 90 image_dir = os.path.join(data_dir, f"{suffix}_DATA", f"{suffix}_images") 91 mask_dir = os.path.join(data_dir, f"{suffix}_DATA", f"{suffix}_masks") 92 93 all_image_paths = sorted(glob(os.path.join(image_dir, f"*_img{suffix}.nii.gz"))) 94 95 image_paths, gt_paths = [], [] 96 for image_path in all_image_paths: 97 case_id = os.path.basename(image_path).replace(f"_img{suffix}.nii.gz", "") 98 if label_choice == "gt_estimated": 99 gt_path = os.path.join(mask_dir, f"GT_estimated_masks{suffix}", f"{case_id}_mask{suffix}.nii.gz") 100 elif label_choice == "annotator1": 101 annotator_id = case_id + "1" if modality == "us" else case_id + "_1" 102 gt_path = os.path.join(mask_dir, "Annotator1", f"{annotator_id}_mask{suffix}.nii.gz") 103 else: 104 annotator_id = case_id + "2" if modality == "us" else case_id + "_2" 105 gt_path = os.path.join(mask_dir, "Annotator2", f"{annotator_id}_mask{suffix}.nii.gz") 106 107 # Not every case has annotations from both annotators, so we skip the ones that are missing. 108 if os.path.exists(gt_path): 109 image_paths.append(image_path) 110 gt_paths.append(gt_path) 111 112 if len(image_paths) == 0 or len(image_paths) != len(gt_paths): 113 raise RuntimeError("Something went wrong with fetching the image and label paths.") 114 115 return image_paths, gt_paths
Get paths to the TRUSTED data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct').
- label_choice: The choice of segmentation annotation to use, either the STAPLE-fused consensus mask ('gt_estimated') or one of the two individual annotators ('annotator1' / 'annotator2').
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
118def get_trusted_dataset( 119 path: Union[os.PathLike, str], 120 patch_shape: Tuple[int, int, int], 121 modality: Literal["us", "ct"] = "us", 122 label_choice: Literal["gt_estimated", "annotator1", "annotator2"] = "gt_estimated", 123 resize_inputs: bool = False, 124 download: bool = False, 125 **kwargs 126) -> Dataset: 127 """Get the TRUSTED dataset for kidney segmentation. 128 129 Args: 130 path: Filepath to a folder where the data is downloaded for further processing. 131 patch_shape: The patch shape to use for training. 132 modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct'). 133 label_choice: The choice of segmentation annotation to use, either the STAPLE-fused 134 consensus mask ('gt_estimated') or one of the two individual annotators 135 ('annotator1' / 'annotator2'). 136 resize_inputs: Whether to resize the inputs. 137 download: Whether to download the data if it is not present. 138 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 139 140 Returns: 141 The segmentation dataset. 142 """ 143 image_paths, gt_paths = get_trusted_paths(path, modality, label_choice, download) 144 145 if resize_inputs: 146 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 147 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 148 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 149 ) 150 151 kwargs.setdefault("pre_label_transform", _binarize_labels) 152 153 return torch_em.default_segmentation_dataset( 154 raw_paths=image_paths, 155 raw_key="data", 156 label_paths=gt_paths, 157 label_key="data", 158 patch_shape=patch_shape, 159 **kwargs 160 )
Get the TRUSTED dataset for kidney segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct').
- label_choice: The choice of segmentation annotation to use, either the STAPLE-fused consensus mask ('gt_estimated') or one of the two individual annotators ('annotator1' / 'annotator2').
- resize_inputs: Whether to resize the inputs.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
163def get_trusted_loader( 164 path: Union[os.PathLike, str], 165 batch_size: int, 166 patch_shape: Tuple[int, int, int], 167 modality: Literal["us", "ct"] = "us", 168 label_choice: Literal["gt_estimated", "annotator1", "annotator2"] = "gt_estimated", 169 resize_inputs: bool = False, 170 download: bool = False, 171 **kwargs 172) -> DataLoader: 173 """Get the TRUSTED dataloader for kidney segmentation. 174 175 Args: 176 path: Filepath to a folder where the data is downloaded for further processing. 177 batch_size: The batch size for training. 178 patch_shape: The patch shape to use for training. 179 modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct'). 180 label_choice: The choice of segmentation annotation to use, either the STAPLE-fused 181 consensus mask ('gt_estimated') or one of the two individual annotators 182 ('annotator1' / 'annotator2'). 183 resize_inputs: Whether to resize the inputs. 184 download: Whether to download the data if it is not present. 185 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 186 187 Returns: 188 The DataLoader. 189 """ 190 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 191 dataset = get_trusted_dataset(path, patch_shape, modality, label_choice, resize_inputs, download, **ds_kwargs) 192 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the TRUSTED dataloader for kidney segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- modality: The choice of imaging modality, either 3d transabdominal ultrasound ('us') or CT ('ct').
- label_choice: The choice of segmentation annotation to use, either the STAPLE-fused consensus mask ('gt_estimated') or one of the two individual annotators ('annotator1' / 'annotator2').
- resize_inputs: Whether to resize the inputs.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.