torch_em.data.datasets.medical.mct_ltdiag
The MCT-LTDiag dataset contains annotations for liver tumor segmentation in multi-phase contrast-enhanced CT.
The dataset consists of 517 patients (one tar archive per patient) with four-phase CT acquisitions
(non-contrast 'nc', arterial 'art', portal-venous 'pvp' and delayed 'delay', stored as
'NIFTI/
The data is located at https://doi.org/10.7910/DVN/S3RW15, released under a CC0-1.0 license, and is downloadable anonymously via the Harvard Dataverse access API (no personal API token required for these unrestricted files, despite the dataset-level metadata suggesting otherwise).
This dataset is from the publication https://doi.org/10.1148/radiol.232214. Please cite it if you use this dataset for your research.
1"""The MCT-LTDiag dataset contains annotations for liver tumor segmentation in multi-phase 2contrast-enhanced CT. 3 4The dataset consists of 517 patients (one tar archive per patient) with four-phase CT acquisitions 5(non-contrast 'nc', arterial 'art', portal-venous 'pvp' and delayed 'delay', stored as 6'NIFTI/<phase>.nii.gz'). Manual tumor and whole-liver segmentation masks ('mask_pvp.nii.gz' and 7'liver_mask_pvp.nii.gz') are only available for the portal-venous phase, which this loader pairs 8with 'NIFTI/pvp.nii.gz'. Each patient is additionally labeled with one of 5 tumor types (e.g. 'BCLM' 9for breast-cancer liver metastasis) in 'meta_info_patient.tab', which is not used by this loader but 10kept alongside the downloaded data for reference. 11 12The data is located at https://doi.org/10.7910/DVN/S3RW15, released under a CC0-1.0 license, and is 13downloadable anonymously via the Harvard Dataverse access API (no personal API token required for 14these unrestricted files, despite the dataset-level metadata suggesting otherwise). 15 16This dataset is from the publication https://doi.org/10.1148/radiol.232214. 17Please cite it if you use this dataset for your research. 18""" 19 20import os 21from glob import glob 22from natsort import natsorted 23from typing import Union, Tuple, Literal, List, Optional 24 25from torch.utils.data import Dataset, DataLoader 26 27import torch_em 28 29from .. import util 30 31 32DATAVERSE_API_URL = "https://dataverse.harvard.edu/api/datasets/:persistentId/?persistentId=doi:10.7910/DVN/S3RW15" 33DATAVERSE_DOWNLOAD_URL = "https://dataverse.harvard.edu/api/access/datafile/{file_id}" 34 35TARGETS = {"liver": "liver_mask_pvp.nii.gz", "tumor": "mask_pvp.nii.gz"} 36 37 38def _list_patient_files(): 39 import requests 40 41 # The default python-requests user agent is blocked by the Dataverse API, unlike a browser-like one. 42 response = requests.get(DATAVERSE_API_URL, headers={"User-Agent": "Mozilla/5.0"}) 43 response.raise_for_status() 44 files = response.json()["data"]["latestVersion"]["files"] 45 return { 46 f["label"][:-len(".tar")]: f["dataFile"]["id"] for f in files if f["label"].endswith(".tar") 47 } 48 49 50def get_mct_ltdiag_data( 51 path: Union[os.PathLike, str], n_patients: Optional[int] = None, download: bool = False 52) -> str: 53 """Download the MCT-LTDiag dataset. 54 55 NOTE: The full collection is about 180 GB. Use `n_patients` to only download a subset for a quick start. 56 57 Args: 58 path: Filepath to a folder where the data is downloaded for further processing. 59 n_patients: The number of patients to download, sorted by patient id. By default all 517 are downloaded. 60 download: Whether to download the data if it is not present. 61 62 Returns: 63 Filepath where the data is downloaded. 64 """ 65 import tarfile 66 67 patient_dir = os.path.join(path, "patients") 68 patient_files = _list_patient_files() 69 patient_ids = sorted(patient_files) 70 if n_patients is not None: 71 patient_ids = patient_ids[:n_patients] 72 missing = [ 73 patient_id for patient_id in patient_ids 74 if not os.path.exists(os.path.join(patient_dir, patient_id, "NIFTI", "pvp.nii.gz")) 75 ] 76 if not missing: 77 return patient_dir 78 79 os.makedirs(patient_dir, exist_ok=True) 80 for patient_id in missing: 81 tar_path = os.path.join(path, f"{patient_id}.tar") 82 url = DATAVERSE_DOWNLOAD_URL.format(file_id=patient_files[patient_id]) 83 util.download_source(path=tar_path, url=url, download=download) 84 85 out_dir = os.path.join(patient_dir, patient_id) 86 os.makedirs(out_dir, exist_ok=True) 87 with tarfile.open(tar_path, "r") as tar: 88 tar.extractall(path=out_dir) 89 os.remove(tar_path) 90 91 return patient_dir 92 93 94def get_mct_ltdiag_paths( 95 path: Union[os.PathLike, str], 96 target: Literal["liver", "tumor"] = "tumor", 97 n_patients: Optional[int] = None, 98 download: bool = False, 99) -> Tuple[List[str], List[str]]: 100 """Get paths to the MCT-LTDiag data. 101 102 Args: 103 path: Filepath to a folder where the data is downloaded for further processing. 104 target: The choice of segmentation target. Either 'liver' or 'tumor'. 105 n_patients: The number of patients to use, sorted by patient id. By default all 517 are used. 106 download: Whether to download the data if it is not present. 107 108 Returns: 109 List of filepaths for the image data. 110 List of filepaths for the label data. 111 """ 112 if target not in TARGETS: 113 raise ValueError(f"'{target}' is not a valid target. Choose one of {list(TARGETS)}.") 114 115 patient_dir = get_mct_ltdiag_data(path, n_patients, download) 116 117 raw_paths = natsorted(glob(os.path.join(patient_dir, "*", "NIFTI", "pvp.nii.gz"))) 118 label_paths = [os.path.join(os.path.dirname(os.path.dirname(p)), TARGETS[target]) for p in raw_paths] 119 120 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 121 assert all(os.path.exists(p) for p in label_paths) 122 123 return raw_paths, label_paths 124 125 126def get_mct_ltdiag_dataset( 127 path: Union[os.PathLike, str], 128 patch_shape: Tuple[int, int, int], 129 target: Literal["liver", "tumor"] = "tumor", 130 n_patients: Optional[int] = None, 131 resize_inputs: bool = False, 132 download: bool = False, 133 **kwargs 134) -> Dataset: 135 """Get the MCT-LTDiag dataset for liver and tumor segmentation in portal-venous phase CT. 136 137 Args: 138 path: Filepath to a folder where the data is downloaded for further processing. 139 patch_shape: The patch shape to use for training. 140 target: The choice of segmentation target. Either 'liver' or 'tumor'. 141 n_patients: The number of patients to use, sorted by patient id. By default all 517 are used. 142 resize_inputs: Whether to resize the inputs to the patch shape. 143 download: Whether to download the data if it is not present. 144 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 145 146 Returns: 147 The segmentation dataset. 148 """ 149 raw_paths, label_paths = get_mct_ltdiag_paths(path, target, n_patients, download) 150 151 if resize_inputs: 152 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 153 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 154 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 155 ) 156 157 return torch_em.default_segmentation_dataset( 158 raw_paths=raw_paths, 159 raw_key="data", 160 label_paths=label_paths, 161 label_key="data", 162 is_seg_dataset=True, 163 patch_shape=patch_shape, 164 ndim=3, 165 **kwargs 166 ) 167 168 169def get_mct_ltdiag_loader( 170 path: Union[os.PathLike, str], 171 batch_size: int, 172 patch_shape: Tuple[int, int, int], 173 target: Literal["liver", "tumor"] = "tumor", 174 n_patients: Optional[int] = None, 175 resize_inputs: bool = False, 176 download: bool = False, 177 **kwargs 178) -> DataLoader: 179 """Get the MCT-LTDiag dataloader for liver and tumor segmentation in portal-venous phase CT. 180 181 Args: 182 path: Filepath to a folder where the data is downloaded for further processing. 183 batch_size: The batch size for training. 184 patch_shape: The patch shape to use for training. 185 target: The choice of segmentation target. Either 'liver' or 'tumor'. 186 n_patients: The number of patients to use, sorted by patient id. By default all 517 are used. 187 resize_inputs: Whether to resize the inputs to the patch shape. 188 download: Whether to download the data if it is not present. 189 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 190 191 Returns: 192 The DataLoader. 193 """ 194 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 195 dataset = get_mct_ltdiag_dataset(path, patch_shape, target, n_patients, resize_inputs, download, **ds_kwargs) 196 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
51def get_mct_ltdiag_data( 52 path: Union[os.PathLike, str], n_patients: Optional[int] = None, download: bool = False 53) -> str: 54 """Download the MCT-LTDiag dataset. 55 56 NOTE: The full collection is about 180 GB. Use `n_patients` to only download a subset for a quick start. 57 58 Args: 59 path: Filepath to a folder where the data is downloaded for further processing. 60 n_patients: The number of patients to download, sorted by patient id. By default all 517 are downloaded. 61 download: Whether to download the data if it is not present. 62 63 Returns: 64 Filepath where the data is downloaded. 65 """ 66 import tarfile 67 68 patient_dir = os.path.join(path, "patients") 69 patient_files = _list_patient_files() 70 patient_ids = sorted(patient_files) 71 if n_patients is not None: 72 patient_ids = patient_ids[:n_patients] 73 missing = [ 74 patient_id for patient_id in patient_ids 75 if not os.path.exists(os.path.join(patient_dir, patient_id, "NIFTI", "pvp.nii.gz")) 76 ] 77 if not missing: 78 return patient_dir 79 80 os.makedirs(patient_dir, exist_ok=True) 81 for patient_id in missing: 82 tar_path = os.path.join(path, f"{patient_id}.tar") 83 url = DATAVERSE_DOWNLOAD_URL.format(file_id=patient_files[patient_id]) 84 util.download_source(path=tar_path, url=url, download=download) 85 86 out_dir = os.path.join(patient_dir, patient_id) 87 os.makedirs(out_dir, exist_ok=True) 88 with tarfile.open(tar_path, "r") as tar: 89 tar.extractall(path=out_dir) 90 os.remove(tar_path) 91 92 return patient_dir
Download the MCT-LTDiag dataset.
NOTE: The full collection is about 180 GB. Use n_patients to only download a subset for a quick start.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- n_patients: The number of patients to download, sorted by patient id. By default all 517 are downloaded.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
95def get_mct_ltdiag_paths( 96 path: Union[os.PathLike, str], 97 target: Literal["liver", "tumor"] = "tumor", 98 n_patients: Optional[int] = None, 99 download: bool = False, 100) -> Tuple[List[str], List[str]]: 101 """Get paths to the MCT-LTDiag data. 102 103 Args: 104 path: Filepath to a folder where the data is downloaded for further processing. 105 target: The choice of segmentation target. Either 'liver' or 'tumor'. 106 n_patients: The number of patients to use, sorted by patient id. By default all 517 are used. 107 download: Whether to download the data if it is not present. 108 109 Returns: 110 List of filepaths for the image data. 111 List of filepaths for the label data. 112 """ 113 if target not in TARGETS: 114 raise ValueError(f"'{target}' is not a valid target. Choose one of {list(TARGETS)}.") 115 116 patient_dir = get_mct_ltdiag_data(path, n_patients, download) 117 118 raw_paths = natsorted(glob(os.path.join(patient_dir, "*", "NIFTI", "pvp.nii.gz"))) 119 label_paths = [os.path.join(os.path.dirname(os.path.dirname(p)), TARGETS[target]) for p in raw_paths] 120 121 assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0 122 assert all(os.path.exists(p) for p in label_paths) 123 124 return raw_paths, label_paths
Get paths to the MCT-LTDiag data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- target: The choice of segmentation target. Either 'liver' or 'tumor'.
- n_patients: The number of patients to use, sorted by patient id. By default all 517 are used.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
127def get_mct_ltdiag_dataset( 128 path: Union[os.PathLike, str], 129 patch_shape: Tuple[int, int, int], 130 target: Literal["liver", "tumor"] = "tumor", 131 n_patients: Optional[int] = None, 132 resize_inputs: bool = False, 133 download: bool = False, 134 **kwargs 135) -> Dataset: 136 """Get the MCT-LTDiag dataset for liver and tumor segmentation in portal-venous phase CT. 137 138 Args: 139 path: Filepath to a folder where the data is downloaded for further processing. 140 patch_shape: The patch shape to use for training. 141 target: The choice of segmentation target. Either 'liver' or 'tumor'. 142 n_patients: The number of patients to use, sorted by patient id. By default all 517 are used. 143 resize_inputs: Whether to resize the inputs to the patch shape. 144 download: Whether to download the data if it is not present. 145 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 146 147 Returns: 148 The segmentation dataset. 149 """ 150 raw_paths, label_paths = get_mct_ltdiag_paths(path, target, n_patients, download) 151 152 if resize_inputs: 153 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 154 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 155 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 156 ) 157 158 return torch_em.default_segmentation_dataset( 159 raw_paths=raw_paths, 160 raw_key="data", 161 label_paths=label_paths, 162 label_key="data", 163 is_seg_dataset=True, 164 patch_shape=patch_shape, 165 ndim=3, 166 **kwargs 167 )
Get the MCT-LTDiag dataset for liver and tumor segmentation in portal-venous phase CT.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- target: The choice of segmentation target. Either 'liver' or 'tumor'.
- n_patients: The number of patients to use, sorted by patient id. By default all 517 are used.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
170def get_mct_ltdiag_loader( 171 path: Union[os.PathLike, str], 172 batch_size: int, 173 patch_shape: Tuple[int, int, int], 174 target: Literal["liver", "tumor"] = "tumor", 175 n_patients: Optional[int] = None, 176 resize_inputs: bool = False, 177 download: bool = False, 178 **kwargs 179) -> DataLoader: 180 """Get the MCT-LTDiag dataloader for liver and tumor segmentation in portal-venous phase CT. 181 182 Args: 183 path: Filepath to a folder where the data is downloaded for further processing. 184 batch_size: The batch size for training. 185 patch_shape: The patch shape to use for training. 186 target: The choice of segmentation target. Either 'liver' or 'tumor'. 187 n_patients: The number of patients to use, sorted by patient id. By default all 517 are used. 188 resize_inputs: Whether to resize the inputs to the patch shape. 189 download: Whether to download the data if it is not present. 190 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 191 192 Returns: 193 The DataLoader. 194 """ 195 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 196 dataset = get_mct_ltdiag_dataset(path, patch_shape, target, n_patients, resize_inputs, download, **ds_kwargs) 197 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the MCT-LTDiag dataloader for liver and tumor segmentation in portal-venous phase CT.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- target: The choice of segmentation target. Either 'liver' or 'tumor'.
- n_patients: The number of patients to use, sorted by patient id. By default all 517 are used.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.