torch_em.data.datasets.medical.npc_mri
The NPC MRI dataset contains annotations for tumor segmentation in nasopharyngeal carcinoma (NPC) MRI.
The dataset consists of 831 MRI scans (T1-weighted, T2-weighted and contrast-enhanced T1-weighted axial series) of 277 untreated primary NPC patients, together with expert radiologist tumor segmentations delineating the gross tumor volume. The scans are distributed as DICOM series and the segmentations as binary NIfTI masks (one per patient and modality). The dataset also ships clinical and laboratory metadata (TNM staging, EBV-DNA concentration, biopsy results, survival), which are not exposed by this module.
The dataset is located at https://zenodo.org/records/13131827 (CC BY 4.0). This dataset is from the publication https://doi.org/10.1038/s41597-025-05815-x. Please cite it if you use this dataset for your research.
1"""The NPC MRI dataset contains annotations for tumor segmentation in nasopharyngeal 2carcinoma (NPC) MRI. 3 4The dataset consists of 831 MRI scans (T1-weighted, T2-weighted and contrast-enhanced T1-weighted 5axial series) of 277 untreated primary NPC patients, together with expert radiologist tumor 6segmentations delineating the gross tumor volume. The scans are distributed as DICOM series and the 7segmentations as binary NIfTI masks (one per patient and modality). The dataset also ships clinical 8and laboratory metadata (TNM staging, EBV-DNA concentration, biopsy results, survival), which are 9not exposed by this module. 10 11The dataset is located at https://zenodo.org/records/13131827 (CC BY 4.0). 12This dataset is from the publication https://doi.org/10.1038/s41597-025-05815-x. 13Please cite it if you use this dataset for your research. 14""" 15 16import os 17from glob import glob 18from tqdm import tqdm 19from natsort import natsorted 20from typing import Union, Tuple, Literal, List 21 22import numpy as np 23 24from torch.utils.data import Dataset, DataLoader 25 26import torch_em 27 28from .. import util 29 30 31URL = "https://zenodo.org/records/13131827/files/primary_data.zip" 32CHECKSUM = "7ed81ca69c92a289ed0d6b8d1387bfe3db4ebe1717750cf94721f992f9ae6b4f" 33 34MODALITIES = ["T1", "T2", "CE-T1"] 35 36 37def _preprocess_npc_mri(data_root, preprocessed_dir, modality): 38 import h5py 39 import nibabel as nib 40 41 os.makedirs(preprocessed_dir, exist_ok=True) 42 43 patient_dirs = natsorted(glob(os.path.join(data_root, "*"))) 44 for patient_dir in tqdm(patient_dirs, desc=f"Preprocess NPC MRI ({modality})"): 45 patient_id = os.path.basename(patient_dir) 46 series_dir = os.path.join(patient_dir, f"{modality}WI") 47 mask_path = os.path.join(patient_dir, f"ROI-{modality}.nii") 48 if not os.path.isdir(series_dir) or not os.path.exists(mask_path): 49 continue 50 51 out_path = os.path.join(preprocessed_dir, f"{patient_id}.h5") 52 if os.path.exists(out_path): 53 continue 54 55 volume, _ = util.load_dicom_series(series_dir) # (z, y, x) 56 labels = np.asarray(nib.load(mask_path).dataobj).T # (x, y, z) -> (z, y, x) 57 labels = (np.round(labels) > 0).astype("uint8") 58 59 if labels.shape != volume.shape: 60 continue 61 62 with h5py.File(out_path, "w") as f: 63 f.create_dataset("raw", data=volume.astype("float32"), compression="gzip") 64 f.create_dataset("labels", data=labels, compression="gzip") 65 66 67def get_npc_mri_data( 68 path: Union[os.PathLike, str], modality: Literal["T1", "T2", "CE-T1"] = "CE-T1", download: bool = False 69) -> str: 70 """Download the NPC MRI dataset. 71 72 Args: 73 path: Filepath to a folder where the data is downloaded for further processing. 74 modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'. 75 download: Whether to download the data if it is not present. 76 77 Returns: 78 Filepath to the folder where the preprocessed (per-patient hdf5) data is stored. 79 """ 80 if modality not in MODALITIES: 81 raise ValueError(f"'{modality}' is not a valid modality. Please choose from {MODALITIES}.") 82 83 preprocessed_dir = os.path.join(path, f"preprocessed_{modality}") 84 if os.path.exists(preprocessed_dir) and len(glob(os.path.join(preprocessed_dir, "*.h5"))) > 0: 85 return preprocessed_dir 86 87 os.makedirs(path, exist_ok=True) 88 89 data_root = os.path.join(path, "data", "MRI-Segments") 90 if not os.path.exists(data_root): 91 zip_path = os.path.join(path, "primary_data.zip") 92 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 93 94 import zipfile 95 with zipfile.ZipFile(zip_path) as f: 96 members = [ 97 m for m in f.namelist() if m.startswith("data/MRI-Segments/") and "__MACOSX" not in m 98 ] 99 f.extractall(path, members=members) 100 os.remove(zip_path) 101 102 _preprocess_npc_mri(data_root, preprocessed_dir, modality) 103 104 return preprocessed_dir 105 106 107def get_npc_mri_paths( 108 path: Union[os.PathLike, str], modality: Literal["T1", "T2", "CE-T1"] = "CE-T1", download: bool = False 109) -> List[str]: 110 """Get paths to the NPC MRI data. 111 112 Args: 113 path: Filepath to a folder where the data is downloaded for further processing. 114 modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'. 115 download: Whether to download the data if it is not present. 116 117 Returns: 118 List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels'). 119 """ 120 preprocessed_dir = get_npc_mri_data(path, modality, download) 121 volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5"))) 122 assert len(volume_paths) > 0, f"Could not find the preprocessed data in '{preprocessed_dir}'." 123 return volume_paths 124 125 126def get_npc_mri_dataset( 127 path: Union[os.PathLike, str], 128 patch_shape: Tuple[int, ...], 129 modality: Literal["T1", "T2", "CE-T1"] = "CE-T1", 130 resize_inputs: bool = False, 131 download: bool = False, 132 **kwargs 133) -> Dataset: 134 """Get the NPC MRI dataset for nasopharyngeal carcinoma tumor segmentation in MRI. 135 136 Args: 137 path: Filepath to a folder where the data is downloaded for further processing. 138 patch_shape: The patch shape to use for training. 139 modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'. 140 resize_inputs: Whether to resize inputs to the desired patch shape. 141 download: Whether to download the data if it is not present. 142 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 143 144 Returns: 145 The segmentation dataset. 146 """ 147 volume_paths = get_npc_mri_paths(path, modality, download) 148 149 if resize_inputs: 150 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 151 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 152 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 153 ) 154 155 return torch_em.default_segmentation_dataset( 156 raw_paths=volume_paths, 157 raw_key="raw", 158 label_paths=volume_paths, 159 label_key="labels", 160 patch_shape=patch_shape, 161 is_seg_dataset=True, 162 **kwargs 163 ) 164 165 166def get_npc_mri_loader( 167 path: Union[os.PathLike, str], 168 batch_size: int, 169 patch_shape: Tuple[int, ...], 170 modality: Literal["T1", "T2", "CE-T1"] = "CE-T1", 171 resize_inputs: bool = False, 172 download: bool = False, 173 **kwargs 174) -> DataLoader: 175 """Get the NPC MRI dataloader for nasopharyngeal carcinoma tumor segmentation in MRI. 176 177 Args: 178 path: Filepath to a folder where the data is downloaded for further processing. 179 batch_size: The batch size for training. 180 patch_shape: The patch shape to use for training. 181 modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'. 182 resize_inputs: Whether to resize inputs to the desired patch shape. 183 download: Whether to download the data if it is not present. 184 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 185 186 Returns: 187 The DataLoader. 188 """ 189 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 190 dataset = get_npc_mri_dataset(path, patch_shape, modality, resize_inputs, download, **ds_kwargs) 191 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
68def get_npc_mri_data( 69 path: Union[os.PathLike, str], modality: Literal["T1", "T2", "CE-T1"] = "CE-T1", download: bool = False 70) -> str: 71 """Download the NPC MRI dataset. 72 73 Args: 74 path: Filepath to a folder where the data is downloaded for further processing. 75 modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'. 76 download: Whether to download the data if it is not present. 77 78 Returns: 79 Filepath to the folder where the preprocessed (per-patient hdf5) data is stored. 80 """ 81 if modality not in MODALITIES: 82 raise ValueError(f"'{modality}' is not a valid modality. Please choose from {MODALITIES}.") 83 84 preprocessed_dir = os.path.join(path, f"preprocessed_{modality}") 85 if os.path.exists(preprocessed_dir) and len(glob(os.path.join(preprocessed_dir, "*.h5"))) > 0: 86 return preprocessed_dir 87 88 os.makedirs(path, exist_ok=True) 89 90 data_root = os.path.join(path, "data", "MRI-Segments") 91 if not os.path.exists(data_root): 92 zip_path = os.path.join(path, "primary_data.zip") 93 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 94 95 import zipfile 96 with zipfile.ZipFile(zip_path) as f: 97 members = [ 98 m for m in f.namelist() if m.startswith("data/MRI-Segments/") and "__MACOSX" not in m 99 ] 100 f.extractall(path, members=members) 101 os.remove(zip_path) 102 103 _preprocess_npc_mri(data_root, preprocessed_dir, modality) 104 105 return preprocessed_dir
Download the NPC MRI dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
- download: Whether to download the data if it is not present.
Returns:
Filepath to the folder where the preprocessed (per-patient hdf5) data is stored.
108def get_npc_mri_paths( 109 path: Union[os.PathLike, str], modality: Literal["T1", "T2", "CE-T1"] = "CE-T1", download: bool = False 110) -> List[str]: 111 """Get paths to the NPC MRI data. 112 113 Args: 114 path: Filepath to a folder where the data is downloaded for further processing. 115 modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'. 116 download: Whether to download the data if it is not present. 117 118 Returns: 119 List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels'). 120 """ 121 preprocessed_dir = get_npc_mri_data(path, modality, download) 122 volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5"))) 123 assert len(volume_paths) > 0, f"Could not find the preprocessed data in '{preprocessed_dir}'." 124 return volume_paths
Get paths to the NPC MRI data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').
127def get_npc_mri_dataset( 128 path: Union[os.PathLike, str], 129 patch_shape: Tuple[int, ...], 130 modality: Literal["T1", "T2", "CE-T1"] = "CE-T1", 131 resize_inputs: bool = False, 132 download: bool = False, 133 **kwargs 134) -> Dataset: 135 """Get the NPC MRI dataset for nasopharyngeal carcinoma tumor segmentation in MRI. 136 137 Args: 138 path: Filepath to a folder where the data is downloaded for further processing. 139 patch_shape: The patch shape to use for training. 140 modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'. 141 resize_inputs: Whether to resize inputs to the desired patch shape. 142 download: Whether to download the data if it is not present. 143 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 144 145 Returns: 146 The segmentation dataset. 147 """ 148 volume_paths = get_npc_mri_paths(path, modality, download) 149 150 if resize_inputs: 151 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 152 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 153 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 154 ) 155 156 return torch_em.default_segmentation_dataset( 157 raw_paths=volume_paths, 158 raw_key="raw", 159 label_paths=volume_paths, 160 label_key="labels", 161 patch_shape=patch_shape, 162 is_seg_dataset=True, 163 **kwargs 164 )
Get the NPC MRI dataset for nasopharyngeal carcinoma tumor segmentation in MRI.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
167def get_npc_mri_loader( 168 path: Union[os.PathLike, str], 169 batch_size: int, 170 patch_shape: Tuple[int, ...], 171 modality: Literal["T1", "T2", "CE-T1"] = "CE-T1", 172 resize_inputs: bool = False, 173 download: bool = False, 174 **kwargs 175) -> DataLoader: 176 """Get the NPC MRI dataloader for nasopharyngeal carcinoma tumor segmentation in MRI. 177 178 Args: 179 path: Filepath to a folder where the data is downloaded for further processing. 180 batch_size: The batch size for training. 181 patch_shape: The patch shape to use for training. 182 modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'. 183 resize_inputs: Whether to resize inputs to the desired patch shape. 184 download: Whether to download the data if it is not present. 185 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 186 187 Returns: 188 The DataLoader. 189 """ 190 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 191 dataset = get_npc_mri_dataset(path, patch_shape, modality, resize_inputs, download, **ds_kwargs) 192 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the NPC MRI dataloader for nasopharyngeal carcinoma tumor segmentation in MRI.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- modality: The choice of MRI modality. One of 'T1', 'T2' or 'CE-T1'.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.