torch_em.data.datasets.medical.npc_mri

The NPC MRI dataset contains annotations for tumor segmentation in nasopharyngeal carcinoma (NPC) MRI.

The dataset consists of 831 MRI scans (T1-weighted, T2-weighted and contrast-enhanced T1-weighted axial series) of 277 untreated primary NPC patients, together with expert radiologist tumor segmentations delineating the gross tumor volume. The scans are distributed as DICOM series and the segmentations as binary NIfTI masks (one per patient and modality). The dataset also ships clinical and laboratory metadata (TNM staging, EBV-DNA concentration, biopsy results, survival), which are not exposed by this module.

The dataset is located at https://zenodo.org/records/13131827 (CC BY 4.0). This dataset is from the publication https://doi.org/10.1038/s41597-025-05815-x. Please cite it if you use this dataset for your research.

  1"""The NPC MRI dataset contains annotations for tumor segmentation in nasopharyngeal
  2carcinoma (NPC) MRI.
  3
  4The dataset consists of 831 MRI scans (T1-weighted, T2-weighted and contrast-enhanced T1-weighted
  5axial series) of 277 untreated primary NPC patients, together with expert radiologist tumor
  6segmentations delineating the gross tumor volume. The scans are distributed as DICOM series and the
  7segmentations as binary NIfTI masks (one per patient and modality). The dataset also ships clinical
  8and laboratory metadata (TNM staging, EBV-DNA concentration, biopsy results, survival), which are
  9not exposed by this module.
 10
 11The dataset is located at https://zenodo.org/records/13131827 (CC BY 4.0).
 12This dataset is from the publication https://doi.org/10.1038/s41597-025-05815-x.
 13Please cite it if you use this dataset for your research.
 14"""
 15
 16import os
 17from glob import glob
 18from tqdm import tqdm
 19from natsort import natsorted
 20from typing import Union, Tuple, Literal, List
 21
 22import numpy as np
 23
 24from torch.utils.data import Dataset, DataLoader
 25
 26import torch_em
 27
 28from .. import util
 29
 30
 31URL = "https://zenodo.org/records/13131827/files/primary_data.zip"
 32CHECKSUM = "7ed81ca69c92a289ed0d6b8d1387bfe3db4ebe1717750cf94721f992f9ae6b4f"
 33
 34MODALITIES = ["T1", "T2", "CE-T1"]
 35
 36
 37def _preprocess_npc_mri(data_root, preprocessed_dir, modality):
 38    import h5py
 39    import nibabel as nib
 40
 41    os.makedirs(preprocessed_dir, exist_ok=True)
 42
 43    patient_dirs = natsorted(glob(os.path.join(data_root, "*")))
 44    for patient_dir in tqdm(patient_dirs, desc=f"Preprocess NPC MRI ({modality})"):
 45        patient_id = os.path.basename(patient_dir)
 46        series_dir = os.path.join(patient_dir, f"{modality}WI")
 47        mask_path = os.path.join(patient_dir, f"ROI-{modality}.nii")
 48        if not os.path.isdir(series_dir) or not os.path.exists(mask_path):
 49            continue
 50
 51        out_path = os.path.join(preprocessed_dir, f"{patient_id}.h5")
 52        if os.path.exists(out_path):
 53            continue
 54
 55        volume, _ = util.load_dicom_series(series_dir)  # (z, y, x)
 56        labels = np.asarray(nib.load(mask_path).dataobj).T  # (x, y, z) -> (z, y, x)
 57        labels = (np.round(labels) > 0).astype("uint8")
 58
 59        if labels.shape != volume.shape:
 60            continue
 61
 62        with h5py.File(out_path, "w") as f:
 63            f.create_dataset("raw", data=volume.astype("float32"), compression="gzip")
 64            f.create_dataset("labels", data=labels, compression="gzip")
 65
 66
 67def get_npc_mri_data(
 68    path: Union[os.PathLike, str], modality: Literal["T1", "T2", "CE-T1"] = "CE-T1", download: bool = False
 69) -> str:
 70    """Download the NPC MRI dataset.
 71
 72    Args:
 73        path: Filepath to a folder where the data is downloaded for further processing.
 74        modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
 75        download: Whether to download the data if it is not present.
 76
 77    Returns:
 78        Filepath to the folder where the preprocessed (per-patient hdf5) data is stored.
 79    """
 80    if modality not in MODALITIES:
 81        raise ValueError(f"'{modality}' is not a valid modality. Please choose from {MODALITIES}.")
 82
 83    preprocessed_dir = os.path.join(path, f"preprocessed_{modality}")
 84    if os.path.exists(preprocessed_dir) and len(glob(os.path.join(preprocessed_dir, "*.h5"))) > 0:
 85        return preprocessed_dir
 86
 87    os.makedirs(path, exist_ok=True)
 88
 89    data_root = os.path.join(path, "data", "MRI-Segments")
 90    if not os.path.exists(data_root):
 91        zip_path = os.path.join(path, "primary_data.zip")
 92        util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 93
 94        import zipfile
 95        with zipfile.ZipFile(zip_path) as f:
 96            members = [
 97                m for m in f.namelist() if m.startswith("data/MRI-Segments/") and "__MACOSX" not in m
 98            ]
 99            f.extractall(path, members=members)
100        os.remove(zip_path)
101
102    _preprocess_npc_mri(data_root, preprocessed_dir, modality)
103
104    return preprocessed_dir
105
106
107def get_npc_mri_paths(
108    path: Union[os.PathLike, str], modality: Literal["T1", "T2", "CE-T1"] = "CE-T1", download: bool = False
109) -> List[str]:
110    """Get paths to the NPC MRI data.
111
112    Args:
113        path: Filepath to a folder where the data is downloaded for further processing.
114        modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
115        download: Whether to download the data if it is not present.
116
117    Returns:
118        List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').
119    """
120    preprocessed_dir = get_npc_mri_data(path, modality, download)
121    volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5")))
122    assert len(volume_paths) > 0, f"Could not find the preprocessed data in '{preprocessed_dir}'."
123    return volume_paths
124
125
126def get_npc_mri_dataset(
127    path: Union[os.PathLike, str],
128    patch_shape: Tuple[int, ...],
129    modality: Literal["T1", "T2", "CE-T1"] = "CE-T1",
130    resize_inputs: bool = False,
131    download: bool = False,
132    **kwargs
133) -> Dataset:
134    """Get the NPC MRI dataset for nasopharyngeal carcinoma tumor segmentation in MRI.
135
136    Args:
137        path: Filepath to a folder where the data is downloaded for further processing.
138        patch_shape: The patch shape to use for training.
139        modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
140        resize_inputs: Whether to resize inputs to the desired patch shape.
141        download: Whether to download the data if it is not present.
142        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
143
144    Returns:
145        The segmentation dataset.
146    """
147    volume_paths = get_npc_mri_paths(path, modality, download)
148
149    if resize_inputs:
150        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
151        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
152            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
153        )
154
155    return torch_em.default_segmentation_dataset(
156        raw_paths=volume_paths,
157        raw_key="raw",
158        label_paths=volume_paths,
159        label_key="labels",
160        patch_shape=patch_shape,
161        is_seg_dataset=True,
162        **kwargs
163    )
164
165
166def get_npc_mri_loader(
167    path: Union[os.PathLike, str],
168    batch_size: int,
169    patch_shape: Tuple[int, ...],
170    modality: Literal["T1", "T2", "CE-T1"] = "CE-T1",
171    resize_inputs: bool = False,
172    download: bool = False,
173    **kwargs
174) -> DataLoader:
175    """Get the NPC MRI dataloader for nasopharyngeal carcinoma tumor segmentation in MRI.
176
177    Args:
178        path: Filepath to a folder where the data is downloaded for further processing.
179        batch_size: The batch size for training.
180        patch_shape: The patch shape to use for training.
181        modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
182        resize_inputs: Whether to resize inputs to the desired patch shape.
183        download: Whether to download the data if it is not present.
184        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
185
186    Returns:
187        The DataLoader.
188    """
189    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
190    dataset = get_npc_mri_dataset(path, patch_shape, modality, resize_inputs, download, **ds_kwargs)
191    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/records/13131827/files/primary_data.zip'
CHECKSUM = '7ed81ca69c92a289ed0d6b8d1387bfe3db4ebe1717750cf94721f992f9ae6b4f'
MODALITIES = ['T1', 'T2', 'CE-T1']
def get_npc_mri_data( path: Union[os.PathLike, str], modality: Literal['T1', 'T2', 'CE-T1'] = 'CE-T1', download: bool = False) -> str:
 68def get_npc_mri_data(
 69    path: Union[os.PathLike, str], modality: Literal["T1", "T2", "CE-T1"] = "CE-T1", download: bool = False
 70) -> str:
 71    """Download the NPC MRI dataset.
 72
 73    Args:
 74        path: Filepath to a folder where the data is downloaded for further processing.
 75        modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
 76        download: Whether to download the data if it is not present.
 77
 78    Returns:
 79        Filepath to the folder where the preprocessed (per-patient hdf5) data is stored.
 80    """
 81    if modality not in MODALITIES:
 82        raise ValueError(f"'{modality}' is not a valid modality. Please choose from {MODALITIES}.")
 83
 84    preprocessed_dir = os.path.join(path, f"preprocessed_{modality}")
 85    if os.path.exists(preprocessed_dir) and len(glob(os.path.join(preprocessed_dir, "*.h5"))) > 0:
 86        return preprocessed_dir
 87
 88    os.makedirs(path, exist_ok=True)
 89
 90    data_root = os.path.join(path, "data", "MRI-Segments")
 91    if not os.path.exists(data_root):
 92        zip_path = os.path.join(path, "primary_data.zip")
 93        util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 94
 95        import zipfile
 96        with zipfile.ZipFile(zip_path) as f:
 97            members = [
 98                m for m in f.namelist() if m.startswith("data/MRI-Segments/") and "__MACOSX" not in m
 99            ]
100            f.extractall(path, members=members)
101        os.remove(zip_path)
102
103    _preprocess_npc_mri(data_root, preprocessed_dir, modality)
104
105    return preprocessed_dir

Download the NPC MRI dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the folder where the preprocessed (per-patient hdf5) data is stored.

def get_npc_mri_paths( path: Union[os.PathLike, str], modality: Literal['T1', 'T2', 'CE-T1'] = 'CE-T1', download: bool = False) -> List[str]:
108def get_npc_mri_paths(
109    path: Union[os.PathLike, str], modality: Literal["T1", "T2", "CE-T1"] = "CE-T1", download: bool = False
110) -> List[str]:
111    """Get paths to the NPC MRI data.
112
113    Args:
114        path: Filepath to a folder where the data is downloaded for further processing.
115        modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
116        download: Whether to download the data if it is not present.
117
118    Returns:
119        List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').
120    """
121    preprocessed_dir = get_npc_mri_data(path, modality, download)
122    volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5")))
123    assert len(volume_paths) > 0, f"Could not find the preprocessed data in '{preprocessed_dir}'."
124    return volume_paths

Get paths to the NPC MRI data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').

def get_npc_mri_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], modality: Literal['T1', 'T2', 'CE-T1'] = 'CE-T1', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
127def get_npc_mri_dataset(
128    path: Union[os.PathLike, str],
129    patch_shape: Tuple[int, ...],
130    modality: Literal["T1", "T2", "CE-T1"] = "CE-T1",
131    resize_inputs: bool = False,
132    download: bool = False,
133    **kwargs
134) -> Dataset:
135    """Get the NPC MRI dataset for nasopharyngeal carcinoma tumor segmentation in MRI.
136
137    Args:
138        path: Filepath to a folder where the data is downloaded for further processing.
139        patch_shape: The patch shape to use for training.
140        modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
141        resize_inputs: Whether to resize inputs to the desired patch shape.
142        download: Whether to download the data if it is not present.
143        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
144
145    Returns:
146        The segmentation dataset.
147    """
148    volume_paths = get_npc_mri_paths(path, modality, download)
149
150    if resize_inputs:
151        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
152        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
153            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
154        )
155
156    return torch_em.default_segmentation_dataset(
157        raw_paths=volume_paths,
158        raw_key="raw",
159        label_paths=volume_paths,
160        label_key="labels",
161        patch_shape=patch_shape,
162        is_seg_dataset=True,
163        **kwargs
164    )

Get the NPC MRI dataset for nasopharyngeal carcinoma tumor segmentation in MRI.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_npc_mri_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], modality: Literal['T1', 'T2', 'CE-T1'] = 'CE-T1', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
167def get_npc_mri_loader(
168    path: Union[os.PathLike, str],
169    batch_size: int,
170    patch_shape: Tuple[int, ...],
171    modality: Literal["T1", "T2", "CE-T1"] = "CE-T1",
172    resize_inputs: bool = False,
173    download: bool = False,
174    **kwargs
175) -> DataLoader:
176    """Get the NPC MRI dataloader for nasopharyngeal carcinoma tumor segmentation in MRI.
177
178    Args:
179        path: Filepath to a folder where the data is downloaded for further processing.
180        batch_size: The batch size for training.
181        patch_shape: The patch shape to use for training.
182        modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
183        resize_inputs: Whether to resize inputs to the desired patch shape.
184        download: Whether to download the data if it is not present.
185        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
186
187    Returns:
188        The DataLoader.
189    """
190    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
191    dataset = get_npc_mri_dataset(path, patch_shape, modality, resize_inputs, download, **ds_kwargs)
192    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the NPC MRI dataloader for nasopharyngeal carcinoma tumor segmentation in MRI.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.