torch_em.data.datasets.medical.pituitary_tumor
The Pituitary-Tumor dataset contains annotations for pituitary neuroendocrine tumor and carotid artery segmentation in contrast-enhanced T1-weighted MRI.
The dataset consists of 136 patients with pituitary adenomas, each with a pre-processed T1-weighted, contrast-enhanced MRI (and a co-registered T2-weighted MRI for most patients), together with manual / semi-automated segmentations of (1) the pituitary tumor-gland complex and (2) the bilateral intracranial carotid arteries.
NOTE: The two binary masks are combined into one semantic label volume with the following ids:
- background: 0
- tumor: 1
- carotids: 2 A few voxels can overlap between the two structures. In this case, the carotids take priority, as they are written after the tumor mask.
NOTE: This requires the nibabel python package.
The dataset is located at https://doi.org/10.6084/m9.figshare.27894084.
This dataset is from the publication https://doi.org/10.1038/s41597-024-04218-8. Please cite it if you use this dataset in your research.
1"""The Pituitary-Tumor dataset contains annotations for pituitary neuroendocrine tumor and 2carotid artery segmentation in contrast-enhanced T1-weighted MRI. 3 4The dataset consists of 136 patients with pituitary adenomas, each with a pre-processed T1-weighted, 5contrast-enhanced MRI (and a co-registered T2-weighted MRI for most patients), together with manual / 6semi-automated segmentations of (1) the pituitary tumor-gland complex and (2) the bilateral intracranial 7carotid arteries. 8 9NOTE: The two binary masks are combined into one semantic label volume with the following ids: 10- background: 0 11- tumor: 1 12- carotids: 2 13A few voxels can overlap between the two structures. In this case, the carotids take priority, as they are 14written after the tumor mask. 15 16NOTE: This requires the nibabel python package. 17 18The dataset is located at https://doi.org/10.6084/m9.figshare.27894084. 19 20This dataset is from the publication https://doi.org/10.1038/s41597-024-04218-8. 21Please cite it if you use this dataset in your research. 22""" 23 24import os 25from glob import glob 26from tqdm import tqdm 27from natsort import natsorted 28from typing import Union, Tuple, List 29 30import numpy as np 31 32from torch.utils.data import Dataset, DataLoader 33 34import torch_em 35 36from .. import util 37 38 39URL = "https://springernature.figshare.com/ndownloader/files/50793273" 40CHECKSUM = "1aaf1945e08083436a7d5a7614c59f07572d0278a211436978a9c3b818cf9074" 41 42LABEL_IDS = {"tumor": 1, "carotids": 2} 43 44# The order in which the structures are written to the label volume. Later structures overwrite earlier ones. 45WRITE_ORDER = ["tumor", "carotids"] 46 47 48def _find_mask_path(t1_path, structure): 49 """Find the mask file for a structure, trying the '_f' (final) suffix used by most patients 50 first, then the plain suffix used by a handful of others. Returns None if neither exists or 51 the only match on disk is an empty (corrupt/missing-annotation) file. 52 """ 53 for suffix in [f"_{structure}_f.nii.gz", f"_{structure}.nii.gz"]: 54 candidate = t1_path.replace("_T1.nii.gz", suffix) 55 if os.path.exists(candidate) and os.path.getsize(candidate) > 0: 56 return candidate 57 return None 58 59 60def _convert_case(t1_path, tumor_path, carotids_path, out_path): 61 import h5py 62 import nibabel as nib 63 from nibabel.processing import resample_from_to 64 65 t1_img = nib.load(t1_path) 66 raw = t1_img.get_fdata() 67 labels = np.zeros(raw.shape, dtype="uint8") 68 for name, mask_path in zip(WRITE_ORDER, [tumor_path, carotids_path]): 69 if mask_path is None: 70 continue 71 mask_img = nib.load(mask_path) 72 if mask_img.shape != raw.shape: 73 # A handful of masks were saved on a slightly different crop of the same (isotropic, 74 # same-spacing) grid as the T1 volume. Resample onto the T1 grid using the affines, 75 # rather than dropping the annotation. 76 mask_img = resample_from_to(mask_img, t1_img, order=0) 77 mask = mask_img.get_fdata() 78 assert mask.shape == raw.shape, f"Shape mismatch for {mask_path}." 79 labels[mask > 0] = LABEL_IDS[name] 80 81 # The nifti volumes are stored in (x, y, z) order, we transpose them to (z, y, x). 82 raw, labels = raw.transpose(2, 1, 0), labels.transpose(2, 1, 0) 83 with h5py.File(out_path, "w") as f: 84 f.create_dataset("raw", data=raw, compression="gzip") 85 f.create_dataset("labels", data=labels, compression="gzip") 86 87 88def get_pituitary_tumor_data(path: Union[os.PathLike, str], download: bool = False) -> str: 89 """Download the Pituitary-Tumor dataset and convert it to hdf5 volumes. 90 91 Args: 92 path: Filepath to a folder where the data is downloaded for further processing. 93 download: Whether to download the data if it is not present. 94 95 Returns: 96 Filepath to the folder with the preprocessed hdf5 volumes. 97 """ 98 data_dir = os.path.join(path, "preprocessed") 99 if os.path.exists(data_dir): 100 return data_dir 101 102 os.makedirs(path, exist_ok=True) 103 104 raw_dir = os.path.join(path, "raw") 105 if not glob(os.path.join(raw_dir, "**", "*_T1.nii.gz"), recursive=True): 106 zip_path = os.path.join(path, "Pituitary_MRI_tumor_carotids.zip") 107 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 108 util.unzip(zip_path=zip_path, dst=raw_dir) 109 110 t1_paths = natsorted(glob(os.path.join(raw_dir, "**", "*_T1.nii.gz"), recursive=True)) 111 assert len(t1_paths) > 0, f"No T1 volumes found at '{raw_dir}'." 112 113 os.makedirs(data_dir, exist_ok=True) 114 for t1_path in tqdm(t1_paths, desc="Converting Pituitary-Tumor volumes to hdf5"): 115 pid = os.path.basename(t1_path).replace("_T1.nii.gz", "") 116 tumor_path = _find_mask_path(t1_path, "tumor") 117 carotids_path = _find_mask_path(t1_path, "carotids") 118 out_path = os.path.join(data_dir, f"{pid}.h5") 119 _convert_case(t1_path, tumor_path, carotids_path, out_path) 120 121 return data_dir 122 123 124def get_pituitary_tumor_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]: 125 """Get paths to the Pituitary-Tumor data. 126 127 Args: 128 path: Filepath to a folder where the data is downloaded for further processing. 129 download: Whether to download the data if it is not present. 130 131 Returns: 132 List of filepaths for the hdf5 volumes with image and label data. 133 """ 134 data_dir = get_pituitary_tumor_data(path, download) 135 volume_paths = natsorted(glob(os.path.join(data_dir, "*.h5"))) 136 assert len(volume_paths) > 0 137 return volume_paths 138 139 140def get_pituitary_tumor_dataset( 141 path: Union[os.PathLike, str], 142 patch_shape: Tuple[int, ...], 143 resize_inputs: bool = False, 144 download: bool = False, 145 **kwargs 146) -> Dataset: 147 """Get the Pituitary-Tumor dataset for pituitary tumor and carotid artery segmentation in MRI. 148 149 Args: 150 path: Filepath to a folder where the data is downloaded for further processing. 151 patch_shape: The patch shape to use for training. 152 resize_inputs: Whether to resize inputs to the desired patch shape. 153 download: Whether to download the data if it is not present. 154 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 155 156 Returns: 157 The segmentation dataset. 158 """ 159 volume_paths = get_pituitary_tumor_paths(path, download) 160 161 if resize_inputs: 162 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 163 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 164 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 165 ) 166 167 return torch_em.default_segmentation_dataset( 168 raw_paths=volume_paths, 169 raw_key="raw", 170 label_paths=volume_paths, 171 label_key="labels", 172 patch_shape=patch_shape, 173 is_seg_dataset=True, 174 **kwargs 175 ) 176 177 178def get_pituitary_tumor_loader( 179 path: Union[os.PathLike, str], 180 batch_size: int, 181 patch_shape: Tuple[int, ...], 182 resize_inputs: bool = False, 183 download: bool = False, 184 **kwargs 185) -> DataLoader: 186 """Get the Pituitary-Tumor dataloader for pituitary tumor and carotid artery segmentation in MRI. 187 188 Args: 189 path: Filepath to a folder where the data is downloaded for further processing. 190 batch_size: The batch size for training. 191 patch_shape: The patch shape to use for training. 192 resize_inputs: Whether to resize inputs to the desired patch shape. 193 download: Whether to download the data if it is not present. 194 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 195 196 Returns: 197 The DataLoader. 198 """ 199 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 200 dataset = get_pituitary_tumor_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 201 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
89def get_pituitary_tumor_data(path: Union[os.PathLike, str], download: bool = False) -> str: 90 """Download the Pituitary-Tumor dataset and convert it to hdf5 volumes. 91 92 Args: 93 path: Filepath to a folder where the data is downloaded for further processing. 94 download: Whether to download the data if it is not present. 95 96 Returns: 97 Filepath to the folder with the preprocessed hdf5 volumes. 98 """ 99 data_dir = os.path.join(path, "preprocessed") 100 if os.path.exists(data_dir): 101 return data_dir 102 103 os.makedirs(path, exist_ok=True) 104 105 raw_dir = os.path.join(path, "raw") 106 if not glob(os.path.join(raw_dir, "**", "*_T1.nii.gz"), recursive=True): 107 zip_path = os.path.join(path, "Pituitary_MRI_tumor_carotids.zip") 108 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 109 util.unzip(zip_path=zip_path, dst=raw_dir) 110 111 t1_paths = natsorted(glob(os.path.join(raw_dir, "**", "*_T1.nii.gz"), recursive=True)) 112 assert len(t1_paths) > 0, f"No T1 volumes found at '{raw_dir}'." 113 114 os.makedirs(data_dir, exist_ok=True) 115 for t1_path in tqdm(t1_paths, desc="Converting Pituitary-Tumor volumes to hdf5"): 116 pid = os.path.basename(t1_path).replace("_T1.nii.gz", "") 117 tumor_path = _find_mask_path(t1_path, "tumor") 118 carotids_path = _find_mask_path(t1_path, "carotids") 119 out_path = os.path.join(data_dir, f"{pid}.h5") 120 _convert_case(t1_path, tumor_path, carotids_path, out_path) 121 122 return data_dir
Download the Pituitary-Tumor dataset and convert it to hdf5 volumes.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath to the folder with the preprocessed hdf5 volumes.
125def get_pituitary_tumor_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]: 126 """Get paths to the Pituitary-Tumor data. 127 128 Args: 129 path: Filepath to a folder where the data is downloaded for further processing. 130 download: Whether to download the data if it is not present. 131 132 Returns: 133 List of filepaths for the hdf5 volumes with image and label data. 134 """ 135 data_dir = get_pituitary_tumor_data(path, download) 136 volume_paths = natsorted(glob(os.path.join(data_dir, "*.h5"))) 137 assert len(volume_paths) > 0 138 return volume_paths
Get paths to the Pituitary-Tumor data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the hdf5 volumes with image and label data.
141def get_pituitary_tumor_dataset( 142 path: Union[os.PathLike, str], 143 patch_shape: Tuple[int, ...], 144 resize_inputs: bool = False, 145 download: bool = False, 146 **kwargs 147) -> Dataset: 148 """Get the Pituitary-Tumor dataset for pituitary tumor and carotid artery segmentation in MRI. 149 150 Args: 151 path: Filepath to a folder where the data is downloaded for further processing. 152 patch_shape: The patch shape to use for training. 153 resize_inputs: Whether to resize inputs to the desired patch shape. 154 download: Whether to download the data if it is not present. 155 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 156 157 Returns: 158 The segmentation dataset. 159 """ 160 volume_paths = get_pituitary_tumor_paths(path, download) 161 162 if resize_inputs: 163 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 164 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 165 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 166 ) 167 168 return torch_em.default_segmentation_dataset( 169 raw_paths=volume_paths, 170 raw_key="raw", 171 label_paths=volume_paths, 172 label_key="labels", 173 patch_shape=patch_shape, 174 is_seg_dataset=True, 175 **kwargs 176 )
Get the Pituitary-Tumor dataset for pituitary tumor and carotid artery segmentation in MRI.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
179def get_pituitary_tumor_loader( 180 path: Union[os.PathLike, str], 181 batch_size: int, 182 patch_shape: Tuple[int, ...], 183 resize_inputs: bool = False, 184 download: bool = False, 185 **kwargs 186) -> DataLoader: 187 """Get the Pituitary-Tumor dataloader for pituitary tumor and carotid artery segmentation in MRI. 188 189 Args: 190 path: Filepath to a folder where the data is downloaded for further processing. 191 batch_size: The batch size for training. 192 patch_shape: The patch shape to use for training. 193 resize_inputs: Whether to resize inputs to the desired patch shape. 194 download: Whether to download the data if it is not present. 195 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 196 197 Returns: 198 The DataLoader. 199 """ 200 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 201 dataset = get_pituitary_tumor_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 202 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the Pituitary-Tumor dataloader for pituitary tumor and carotid artery segmentation in MRI.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.