torch_em.data.datasets.medical.pituitary_tumor

The Pituitary-Tumor dataset contains annotations for pituitary neuroendocrine tumor and carotid artery segmentation in contrast-enhanced T1-weighted MRI.

The dataset consists of 136 patients with pituitary adenomas, each with a pre-processed T1-weighted, contrast-enhanced MRI (and a co-registered T2-weighted MRI for most patients), together with manual / semi-automated segmentations of (1) the pituitary tumor-gland complex and (2) the bilateral intracranial carotid arteries.

NOTE: The two binary masks are combined into one semantic label volume with the following ids:

  • background: 0
  • tumor: 1
  • carotids: 2 A few voxels can overlap between the two structures. In this case, the carotids take priority, as they are written after the tumor mask.

NOTE: This requires the nibabel python package.

The dataset is located at https://doi.org/10.6084/m9.figshare.27894084.

This dataset is from the publication https://doi.org/10.1038/s41597-024-04218-8. Please cite it if you use this dataset in your research.

  1"""The Pituitary-Tumor dataset contains annotations for pituitary neuroendocrine tumor and
  2carotid artery segmentation in contrast-enhanced T1-weighted MRI.
  3
  4The dataset consists of 136 patients with pituitary adenomas, each with a pre-processed T1-weighted,
  5contrast-enhanced MRI (and a co-registered T2-weighted MRI for most patients), together with manual /
  6semi-automated segmentations of (1) the pituitary tumor-gland complex and (2) the bilateral intracranial
  7carotid arteries.
  8
  9NOTE: The two binary masks are combined into one semantic label volume with the following ids:
 10- background: 0
 11- tumor: 1
 12- carotids: 2
 13A few voxels can overlap between the two structures. In this case, the carotids take priority, as they are
 14written after the tumor mask.
 15
 16NOTE: This requires the nibabel python package.
 17
 18The dataset is located at https://doi.org/10.6084/m9.figshare.27894084.
 19
 20This dataset is from the publication https://doi.org/10.1038/s41597-024-04218-8.
 21Please cite it if you use this dataset in your research.
 22"""
 23
 24import os
 25from glob import glob
 26from tqdm import tqdm
 27from natsort import natsorted
 28from typing import Union, Tuple, List
 29
 30import numpy as np
 31
 32from torch.utils.data import Dataset, DataLoader
 33
 34import torch_em
 35
 36from .. import util
 37
 38
 39URL = "https://springernature.figshare.com/ndownloader/files/50793273"
 40CHECKSUM = "1aaf1945e08083436a7d5a7614c59f07572d0278a211436978a9c3b818cf9074"
 41
 42LABEL_IDS = {"tumor": 1, "carotids": 2}
 43
 44# The order in which the structures are written to the label volume. Later structures overwrite earlier ones.
 45WRITE_ORDER = ["tumor", "carotids"]
 46
 47
 48def _find_mask_path(t1_path, structure):
 49    """Find the mask file for a structure, trying the '_f' (final) suffix used by most patients
 50    first, then the plain suffix used by a handful of others. Returns None if neither exists or
 51    the only match on disk is an empty (corrupt/missing-annotation) file.
 52    """
 53    for suffix in [f"_{structure}_f.nii.gz", f"_{structure}.nii.gz"]:
 54        candidate = t1_path.replace("_T1.nii.gz", suffix)
 55        if os.path.exists(candidate) and os.path.getsize(candidate) > 0:
 56            return candidate
 57    return None
 58
 59
 60def _convert_case(t1_path, tumor_path, carotids_path, out_path):
 61    import h5py
 62    import nibabel as nib
 63    from nibabel.processing import resample_from_to
 64
 65    t1_img = nib.load(t1_path)
 66    raw = t1_img.get_fdata()
 67    labels = np.zeros(raw.shape, dtype="uint8")
 68    for name, mask_path in zip(WRITE_ORDER, [tumor_path, carotids_path]):
 69        if mask_path is None:
 70            continue
 71        mask_img = nib.load(mask_path)
 72        if mask_img.shape != raw.shape:
 73            # A handful of masks were saved on a slightly different crop of the same (isotropic,
 74            # same-spacing) grid as the T1 volume. Resample onto the T1 grid using the affines,
 75            # rather than dropping the annotation.
 76            mask_img = resample_from_to(mask_img, t1_img, order=0)
 77        mask = mask_img.get_fdata()
 78        assert mask.shape == raw.shape, f"Shape mismatch for {mask_path}."
 79        labels[mask > 0] = LABEL_IDS[name]
 80
 81    # The nifti volumes are stored in (x, y, z) order, we transpose them to (z, y, x).
 82    raw, labels = raw.transpose(2, 1, 0), labels.transpose(2, 1, 0)
 83    with h5py.File(out_path, "w") as f:
 84        f.create_dataset("raw", data=raw, compression="gzip")
 85        f.create_dataset("labels", data=labels, compression="gzip")
 86
 87
 88def get_pituitary_tumor_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 89    """Download the Pituitary-Tumor dataset and convert it to hdf5 volumes.
 90
 91    Args:
 92        path: Filepath to a folder where the data is downloaded for further processing.
 93        download: Whether to download the data if it is not present.
 94
 95    Returns:
 96        Filepath to the folder with the preprocessed hdf5 volumes.
 97    """
 98    data_dir = os.path.join(path, "preprocessed")
 99    if os.path.exists(data_dir):
100        return data_dir
101
102    os.makedirs(path, exist_ok=True)
103
104    raw_dir = os.path.join(path, "raw")
105    if not glob(os.path.join(raw_dir, "**", "*_T1.nii.gz"), recursive=True):
106        zip_path = os.path.join(path, "Pituitary_MRI_tumor_carotids.zip")
107        util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
108        util.unzip(zip_path=zip_path, dst=raw_dir)
109
110    t1_paths = natsorted(glob(os.path.join(raw_dir, "**", "*_T1.nii.gz"), recursive=True))
111    assert len(t1_paths) > 0, f"No T1 volumes found at '{raw_dir}'."
112
113    os.makedirs(data_dir, exist_ok=True)
114    for t1_path in tqdm(t1_paths, desc="Converting Pituitary-Tumor volumes to hdf5"):
115        pid = os.path.basename(t1_path).replace("_T1.nii.gz", "")
116        tumor_path = _find_mask_path(t1_path, "tumor")
117        carotids_path = _find_mask_path(t1_path, "carotids")
118        out_path = os.path.join(data_dir, f"{pid}.h5")
119        _convert_case(t1_path, tumor_path, carotids_path, out_path)
120
121    return data_dir
122
123
124def get_pituitary_tumor_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
125    """Get paths to the Pituitary-Tumor data.
126
127    Args:
128        path: Filepath to a folder where the data is downloaded for further processing.
129        download: Whether to download the data if it is not present.
130
131    Returns:
132        List of filepaths for the hdf5 volumes with image and label data.
133    """
134    data_dir = get_pituitary_tumor_data(path, download)
135    volume_paths = natsorted(glob(os.path.join(data_dir, "*.h5")))
136    assert len(volume_paths) > 0
137    return volume_paths
138
139
140def get_pituitary_tumor_dataset(
141    path: Union[os.PathLike, str],
142    patch_shape: Tuple[int, ...],
143    resize_inputs: bool = False,
144    download: bool = False,
145    **kwargs
146) -> Dataset:
147    """Get the Pituitary-Tumor dataset for pituitary tumor and carotid artery segmentation in MRI.
148
149    Args:
150        path: Filepath to a folder where the data is downloaded for further processing.
151        patch_shape: The patch shape to use for training.
152        resize_inputs: Whether to resize inputs to the desired patch shape.
153        download: Whether to download the data if it is not present.
154        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
155
156    Returns:
157        The segmentation dataset.
158    """
159    volume_paths = get_pituitary_tumor_paths(path, download)
160
161    if resize_inputs:
162        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
163        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
164            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
165        )
166
167    return torch_em.default_segmentation_dataset(
168        raw_paths=volume_paths,
169        raw_key="raw",
170        label_paths=volume_paths,
171        label_key="labels",
172        patch_shape=patch_shape,
173        is_seg_dataset=True,
174        **kwargs
175    )
176
177
178def get_pituitary_tumor_loader(
179    path: Union[os.PathLike, str],
180    batch_size: int,
181    patch_shape: Tuple[int, ...],
182    resize_inputs: bool = False,
183    download: bool = False,
184    **kwargs
185) -> DataLoader:
186    """Get the Pituitary-Tumor dataloader for pituitary tumor and carotid artery segmentation in MRI.
187
188    Args:
189        path: Filepath to a folder where the data is downloaded for further processing.
190        batch_size: The batch size for training.
191        patch_shape: The patch shape to use for training.
192        resize_inputs: Whether to resize inputs to the desired patch shape.
193        download: Whether to download the data if it is not present.
194        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
195
196    Returns:
197        The DataLoader.
198    """
199    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
200    dataset = get_pituitary_tumor_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
201    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://springernature.figshare.com/ndownloader/files/50793273'
CHECKSUM = '1aaf1945e08083436a7d5a7614c59f07572d0278a211436978a9c3b818cf9074'
LABEL_IDS = {'tumor': 1, 'carotids': 2}
WRITE_ORDER = ['tumor', 'carotids']
def get_pituitary_tumor_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 89def get_pituitary_tumor_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 90    """Download the Pituitary-Tumor dataset and convert it to hdf5 volumes.
 91
 92    Args:
 93        path: Filepath to a folder where the data is downloaded for further processing.
 94        download: Whether to download the data if it is not present.
 95
 96    Returns:
 97        Filepath to the folder with the preprocessed hdf5 volumes.
 98    """
 99    data_dir = os.path.join(path, "preprocessed")
100    if os.path.exists(data_dir):
101        return data_dir
102
103    os.makedirs(path, exist_ok=True)
104
105    raw_dir = os.path.join(path, "raw")
106    if not glob(os.path.join(raw_dir, "**", "*_T1.nii.gz"), recursive=True):
107        zip_path = os.path.join(path, "Pituitary_MRI_tumor_carotids.zip")
108        util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
109        util.unzip(zip_path=zip_path, dst=raw_dir)
110
111    t1_paths = natsorted(glob(os.path.join(raw_dir, "**", "*_T1.nii.gz"), recursive=True))
112    assert len(t1_paths) > 0, f"No T1 volumes found at '{raw_dir}'."
113
114    os.makedirs(data_dir, exist_ok=True)
115    for t1_path in tqdm(t1_paths, desc="Converting Pituitary-Tumor volumes to hdf5"):
116        pid = os.path.basename(t1_path).replace("_T1.nii.gz", "")
117        tumor_path = _find_mask_path(t1_path, "tumor")
118        carotids_path = _find_mask_path(t1_path, "carotids")
119        out_path = os.path.join(data_dir, f"{pid}.h5")
120        _convert_case(t1_path, tumor_path, carotids_path, out_path)
121
122    return data_dir

Download the Pituitary-Tumor dataset and convert it to hdf5 volumes.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the folder with the preprocessed hdf5 volumes.

def get_pituitary_tumor_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
125def get_pituitary_tumor_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
126    """Get paths to the Pituitary-Tumor data.
127
128    Args:
129        path: Filepath to a folder where the data is downloaded for further processing.
130        download: Whether to download the data if it is not present.
131
132    Returns:
133        List of filepaths for the hdf5 volumes with image and label data.
134    """
135    data_dir = get_pituitary_tumor_data(path, download)
136    volume_paths = natsorted(glob(os.path.join(data_dir, "*.h5")))
137    assert len(volume_paths) > 0
138    return volume_paths

Get paths to the Pituitary-Tumor data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the hdf5 volumes with image and label data.

def get_pituitary_tumor_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
141def get_pituitary_tumor_dataset(
142    path: Union[os.PathLike, str],
143    patch_shape: Tuple[int, ...],
144    resize_inputs: bool = False,
145    download: bool = False,
146    **kwargs
147) -> Dataset:
148    """Get the Pituitary-Tumor dataset for pituitary tumor and carotid artery segmentation in MRI.
149
150    Args:
151        path: Filepath to a folder where the data is downloaded for further processing.
152        patch_shape: The patch shape to use for training.
153        resize_inputs: Whether to resize inputs to the desired patch shape.
154        download: Whether to download the data if it is not present.
155        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
156
157    Returns:
158        The segmentation dataset.
159    """
160    volume_paths = get_pituitary_tumor_paths(path, download)
161
162    if resize_inputs:
163        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
164        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
165            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
166        )
167
168    return torch_em.default_segmentation_dataset(
169        raw_paths=volume_paths,
170        raw_key="raw",
171        label_paths=volume_paths,
172        label_key="labels",
173        patch_shape=patch_shape,
174        is_seg_dataset=True,
175        **kwargs
176    )

Get the Pituitary-Tumor dataset for pituitary tumor and carotid artery segmentation in MRI.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_pituitary_tumor_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
179def get_pituitary_tumor_loader(
180    path: Union[os.PathLike, str],
181    batch_size: int,
182    patch_shape: Tuple[int, ...],
183    resize_inputs: bool = False,
184    download: bool = False,
185    **kwargs
186) -> DataLoader:
187    """Get the Pituitary-Tumor dataloader for pituitary tumor and carotid artery segmentation in MRI.
188
189    Args:
190        path: Filepath to a folder where the data is downloaded for further processing.
191        batch_size: The batch size for training.
192        patch_shape: The patch shape to use for training.
193        resize_inputs: Whether to resize inputs to the desired patch shape.
194        download: Whether to download the data if it is not present.
195        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
196
197    Returns:
198        The DataLoader.
199    """
200    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
201    dataset = get_pituitary_tumor_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
202    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the Pituitary-Tumor dataloader for pituitary tumor and carotid artery segmentation in MRI.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.