torch_em.data.datasets.medical.cptac_pda_tumor
The CPTAC-PDA-Tumor-Annotations dataset contains annotations for pancreatic ductal adenocarcinoma and its metastatic lesions in CT.
The dataset consists of contrast-enhanced CT or MR series of pancreatic cancer patients, with expert contours for the primary tumor and, where present, metastatic lesions (lymph nodes, liver, lung and other sites), each drawn as a separate RTSTRUCT series and paired here with the exact series it references. Every lesion is assigned label id 1, regardless of its anatomical site.
NOTE: This requires the pydicom python package.
The dataset is located at https://doi.org/10.7937/BW9V-BX61 and is distributed under the CC BY 4.0 license. This dataset is from the CPTAC-PDA collection on The Cancer Imaging Archive; please cite the DOI above if you use this dataset in your research.
1"""The CPTAC-PDA-Tumor-Annotations dataset contains annotations for pancreatic ductal adenocarcinoma 2and its metastatic lesions in CT. 3 4The dataset consists of contrast-enhanced CT or MR series of pancreatic cancer patients, with expert 5contours for the primary tumor and, where present, metastatic lesions (lymph nodes, liver, lung and 6other sites), each drawn as a separate RTSTRUCT series and paired here with the exact series it 7references. Every lesion is assigned label id 1, regardless of its anatomical site. 8 9NOTE: This requires the pydicom python package. 10 11The dataset is located at https://doi.org/10.7937/BW9V-BX61 and is distributed under the 12CC BY 4.0 license. 13This dataset is from the CPTAC-PDA collection on The Cancer Imaging Archive; please cite the DOI 14above if you use this dataset in your research. 15""" 16 17import os 18import json 19from glob import glob 20from tqdm import tqdm 21from natsort import natsorted 22from typing import Union, Tuple, List 23 24from torch.utils.data import Dataset, DataLoader 25 26import torch_em 27 28from .. import util 29 30 31COLLECTION = "CPTAC-PDA" 32 33 34def _get_series_metadata(path, download): 35 """Get the metadata of all series in the collection from the NBIA REST API.""" 36 import requests 37 38 metadata_path = os.path.join(path, "cptac_pda_tumor_series.json") 39 if not os.path.exists(metadata_path): 40 if not download: 41 raise RuntimeError(f"Cannot find the data at {path}, but download was set to False.") 42 response = requests.get(f"{util.NBIA_API_URL}getSeries", params={"Collection": COLLECTION}) 43 response.raise_for_status() 44 with open(metadata_path, "w") as f: 45 json.dump(response.json(), f, indent=2) 46 47 with open(metadata_path, "r") as f: 48 return json.load(f) 49 50 51def _is_lesion_rtstruct(series): 52 description = (series.get("SeriesDescription") or "").upper() 53 return series.get("Modality") == "RTSTRUCT" and "SEED POINT" not in description 54 55 56def _referenced_series_uid(rtstruct): 57 referenced_study = rtstruct.ReferencedFrameOfReferenceSequence[0].RTReferencedStudySequence[0] 58 return str(referenced_study.RTReferencedSeriesSequence[0].SeriesInstanceUID) 59 60 61def _preprocess_cptac_pda_tumor(dicom_dir, series_metadata, preprocessed_dir): 62 import h5py 63 import pydicom 64 65 rtstruct_series = [series for series in series_metadata if _is_lesion_rtstruct(series)] 66 67 os.makedirs(preprocessed_dir, exist_ok=True) 68 for series in tqdm(rtstruct_series, desc="Preprocess CPTAC-PDA-Tumor-Annotations"): 69 rtstruct_dir = os.path.join(dicom_dir, series["SeriesInstanceUID"]) 70 rtstruct_paths = glob(os.path.join(rtstruct_dir, "*.dcm")) 71 if not rtstruct_paths: 72 continue 73 74 rtstruct = pydicom.dcmread(rtstruct_paths[0], stop_before_pixels=True) 75 image_uid = _referenced_series_uid(rtstruct) 76 image_dir = os.path.join(dicom_dir, image_uid) 77 if not glob(os.path.join(image_dir, "*.dcm")): 78 continue 79 80 out_path = os.path.join(preprocessed_dir, f"{series['SeriesInstanceUID']}.h5") 81 if os.path.exists(out_path): 82 continue 83 84 volume, geometry = util.load_dicom_series(image_dir) 85 labels = util.rasterize_rtstruct(rtstruct_paths[0], geometry, volume.shape, roi_labels=lambda num, name: 1) 86 if labels.max() == 0: # A few RTSTRUCT files carry no findings. 87 continue 88 89 with h5py.File(out_path, "w") as f: 90 f.create_dataset("raw", data=volume, compression="gzip") 91 f.create_dataset("labels", data=labels, compression="gzip") 92 93 94def get_cptac_pda_tumor_data(path: Union[os.PathLike, str], download: bool = False) -> str: 95 """Download the CPTAC-PDA-Tumor-Annotations dataset. 96 97 Args: 98 path: Filepath to a folder where the data is downloaded for further processing. 99 download: Whether to download the data if it is not present. 100 101 Returns: 102 Filepath where the preprocessed data is stored. 103 """ 104 # NOTE: The preprocessing below skips volumes that were converted already, so an interrupted run resumes. 105 preprocessed_dir = os.path.join(path, "preprocessed") 106 107 os.makedirs(path, exist_ok=True) 108 series_metadata = _get_series_metadata(path, download) 109 110 rtstruct_uids = [series["SeriesInstanceUID"] for series in series_metadata if _is_lesion_rtstruct(series)] 111 112 dicom_dir = os.path.join(path, "dicom") 113 if download: # The RTSTRUCT series are downloaded first, so the image series they reference can be found. 114 util.download_tcia_series( 115 rtstruct_uids, dst=dicom_dir, csv_filename=os.path.join(path, "cptac_pda_tumor_rtstruct") 116 ) 117 118 import pydicom 119 image_uids = set() 120 for uid in rtstruct_uids: 121 rtstruct_paths = glob(os.path.join(dicom_dir, uid, "*.dcm")) 122 if rtstruct_paths: 123 rtstruct = pydicom.dcmread(rtstruct_paths[0], stop_before_pixels=True) 124 image_uids.add(_referenced_series_uid(rtstruct)) 125 util.download_tcia_series( 126 sorted(image_uids), dst=dicom_dir, csv_filename=os.path.join(path, "cptac_pda_tumor_image") 127 ) 128 elif not glob(os.path.join(dicom_dir, "*", "*.dcm")): 129 raise RuntimeError(f"Cannot find the data at {path}, but download was set to False.") 130 131 _preprocess_cptac_pda_tumor(dicom_dir, series_metadata, preprocessed_dir) 132 return preprocessed_dir 133 134 135def get_cptac_pda_tumor_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]: 136 """Get paths to the CPTAC-PDA-Tumor-Annotations data. 137 138 Args: 139 path: Filepath to a folder where the data is downloaded for further processing. 140 download: Whether to download the data if it is not present. 141 142 Returns: 143 List of filepaths for the stored data. 144 """ 145 preprocessed_dir = get_cptac_pda_tumor_data(path, download) 146 volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5"))) 147 assert len(volume_paths) > 0, f"Could not find any preprocessed volumes in '{preprocessed_dir}'." 148 return volume_paths 149 150 151def get_cptac_pda_tumor_dataset( 152 path: Union[os.PathLike, str], 153 patch_shape: Tuple[int, ...], 154 resize_inputs: bool = False, 155 download: bool = False, 156 **kwargs 157) -> Dataset: 158 """Get the CPTAC-PDA-Tumor-Annotations dataset for pancreatic tumor and metastasis segmentation. 159 160 Args: 161 path: Filepath to a folder where the data is downloaded for further processing. 162 patch_shape: The patch shape to use for training. 163 resize_inputs: Whether to resize inputs to the desired patch shape. 164 download: Whether to download the data if it is not present. 165 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 166 167 Returns: 168 The segmentation dataset. 169 """ 170 volume_paths = get_cptac_pda_tumor_paths(path, download) 171 172 if resize_inputs: 173 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 174 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 175 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 176 ) 177 178 return torch_em.default_segmentation_dataset( 179 raw_paths=volume_paths, 180 raw_key="raw", 181 label_paths=volume_paths, 182 label_key="labels", 183 patch_shape=patch_shape, 184 is_seg_dataset=True, 185 **kwargs 186 ) 187 188 189def get_cptac_pda_tumor_loader( 190 path: Union[os.PathLike, str], 191 batch_size: int, 192 patch_shape: Tuple[int, ...], 193 resize_inputs: bool = False, 194 download: bool = False, 195 **kwargs 196) -> DataLoader: 197 """Get the CPTAC-PDA-Tumor-Annotations dataloader for pancreatic tumor and metastasis segmentation. 198 199 Args: 200 path: Filepath to a folder where the data is downloaded for further processing. 201 batch_size: The batch size for training. 202 patch_shape: The patch shape to use for training. 203 resize_inputs: Whether to resize inputs to the desired patch shape. 204 download: Whether to download the data if it is not present. 205 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 206 207 Returns: 208 The DataLoader. 209 """ 210 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 211 dataset = get_cptac_pda_tumor_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 212 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
95def get_cptac_pda_tumor_data(path: Union[os.PathLike, str], download: bool = False) -> str: 96 """Download the CPTAC-PDA-Tumor-Annotations dataset. 97 98 Args: 99 path: Filepath to a folder where the data is downloaded for further processing. 100 download: Whether to download the data if it is not present. 101 102 Returns: 103 Filepath where the preprocessed data is stored. 104 """ 105 # NOTE: The preprocessing below skips volumes that were converted already, so an interrupted run resumes. 106 preprocessed_dir = os.path.join(path, "preprocessed") 107 108 os.makedirs(path, exist_ok=True) 109 series_metadata = _get_series_metadata(path, download) 110 111 rtstruct_uids = [series["SeriesInstanceUID"] for series in series_metadata if _is_lesion_rtstruct(series)] 112 113 dicom_dir = os.path.join(path, "dicom") 114 if download: # The RTSTRUCT series are downloaded first, so the image series they reference can be found. 115 util.download_tcia_series( 116 rtstruct_uids, dst=dicom_dir, csv_filename=os.path.join(path, "cptac_pda_tumor_rtstruct") 117 ) 118 119 import pydicom 120 image_uids = set() 121 for uid in rtstruct_uids: 122 rtstruct_paths = glob(os.path.join(dicom_dir, uid, "*.dcm")) 123 if rtstruct_paths: 124 rtstruct = pydicom.dcmread(rtstruct_paths[0], stop_before_pixels=True) 125 image_uids.add(_referenced_series_uid(rtstruct)) 126 util.download_tcia_series( 127 sorted(image_uids), dst=dicom_dir, csv_filename=os.path.join(path, "cptac_pda_tumor_image") 128 ) 129 elif not glob(os.path.join(dicom_dir, "*", "*.dcm")): 130 raise RuntimeError(f"Cannot find the data at {path}, but download was set to False.") 131 132 _preprocess_cptac_pda_tumor(dicom_dir, series_metadata, preprocessed_dir) 133 return preprocessed_dir
Download the CPTAC-PDA-Tumor-Annotations dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the preprocessed data is stored.
136def get_cptac_pda_tumor_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]: 137 """Get paths to the CPTAC-PDA-Tumor-Annotations data. 138 139 Args: 140 path: Filepath to a folder where the data is downloaded for further processing. 141 download: Whether to download the data if it is not present. 142 143 Returns: 144 List of filepaths for the stored data. 145 """ 146 preprocessed_dir = get_cptac_pda_tumor_data(path, download) 147 volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5"))) 148 assert len(volume_paths) > 0, f"Could not find any preprocessed volumes in '{preprocessed_dir}'." 149 return volume_paths
Get paths to the CPTAC-PDA-Tumor-Annotations data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the stored data.
152def get_cptac_pda_tumor_dataset( 153 path: Union[os.PathLike, str], 154 patch_shape: Tuple[int, ...], 155 resize_inputs: bool = False, 156 download: bool = False, 157 **kwargs 158) -> Dataset: 159 """Get the CPTAC-PDA-Tumor-Annotations dataset for pancreatic tumor and metastasis segmentation. 160 161 Args: 162 path: Filepath to a folder where the data is downloaded for further processing. 163 patch_shape: The patch shape to use for training. 164 resize_inputs: Whether to resize inputs to the desired patch shape. 165 download: Whether to download the data if it is not present. 166 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 167 168 Returns: 169 The segmentation dataset. 170 """ 171 volume_paths = get_cptac_pda_tumor_paths(path, download) 172 173 if resize_inputs: 174 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 175 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 176 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 177 ) 178 179 return torch_em.default_segmentation_dataset( 180 raw_paths=volume_paths, 181 raw_key="raw", 182 label_paths=volume_paths, 183 label_key="labels", 184 patch_shape=patch_shape, 185 is_seg_dataset=True, 186 **kwargs 187 )
Get the CPTAC-PDA-Tumor-Annotations dataset for pancreatic tumor and metastasis segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
190def get_cptac_pda_tumor_loader( 191 path: Union[os.PathLike, str], 192 batch_size: int, 193 patch_shape: Tuple[int, ...], 194 resize_inputs: bool = False, 195 download: bool = False, 196 **kwargs 197) -> DataLoader: 198 """Get the CPTAC-PDA-Tumor-Annotations dataloader for pancreatic tumor and metastasis segmentation. 199 200 Args: 201 path: Filepath to a folder where the data is downloaded for further processing. 202 batch_size: The batch size for training. 203 patch_shape: The patch shape to use for training. 204 resize_inputs: Whether to resize inputs to the desired patch shape. 205 download: Whether to download the data if it is not present. 206 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 207 208 Returns: 209 The DataLoader. 210 """ 211 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 212 dataset = get_cptac_pda_tumor_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 213 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the CPTAC-PDA-Tumor-Annotations dataloader for pancreatic tumor and metastasis segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.