torch_em.data.datasets.medical.lctsc
The LCTSC dataset (Lung CT Segmentation Challenge 2017) contains annotations for lung and thoracic organ segmentation in pretreatment CT of non-small cell lung cancer patients.
It consists of 60 CT volumes with manual delineations by an expert, which are distributed as DICOM
RTSTRUCT contours. This module rasterizes the contours onto the CT grid (see
torch_em.data.datasets.util.rasterize_rtstruct) and stores CT and labels in hdf5 files.
The semantic label ids are: 1: left lung ('Lung_L'), 2: right lung ('Lung_R'), 3: esophagus
('Esophagus'), 4: heart ('Heart'), 5: spinal cord ('SpinalCord'). All five structures are
annotated for every patient.
NOTE: This requires the pydicom python package.
The dataset is located at https://www.cancerimagingarchive.net/collection/lctsc/.
This dataset is from the publication https://doi.org/10.1002/mp.13141. The data was released at https://doi.org/10.7937/K9/TCIA.2017.3R3FVZ08. Please cite it if you use this dataset in your research.
1"""The LCTSC dataset (Lung CT Segmentation Challenge 2017) contains annotations for lung and thoracic 2organ segmentation in pretreatment CT of non-small cell lung cancer patients. 3 4It consists of 60 CT volumes with manual delineations by an expert, which are distributed as DICOM 5RTSTRUCT contours. This module rasterizes the contours onto the CT grid (see 6`torch_em.data.datasets.util.rasterize_rtstruct`) and stores CT and labels in hdf5 files. 7The semantic label ids are: 1: left lung ('Lung_L'), 2: right lung ('Lung_R'), 3: esophagus 8('Esophagus'), 4: heart ('Heart'), 5: spinal cord ('SpinalCord'). All five structures are 9annotated for every patient. 10 11NOTE: This requires the pydicom python package. 12 13The dataset is located at https://www.cancerimagingarchive.net/collection/lctsc/. 14 15This dataset is from the publication https://doi.org/10.1002/mp.13141. 16The data was released at https://doi.org/10.7937/K9/TCIA.2017.3R3FVZ08. 17Please cite it if you use this dataset in your research. 18""" 19 20import os 21import csv 22from glob import glob 23from tqdm import tqdm 24from natsort import natsorted 25from typing import Union, Tuple, List, Optional, Sequence 26 27import numpy as np 28 29from torch.utils.data import Dataset, DataLoader 30 31import torch_em 32from torch_em.transform.generic import Compose 33 34from .. import util 35 36 37URL = "https://www.cancerimagingarchive.net/wp-content/uploads/LCTSC_v2_20190508.tcia" 38 39# The DICOM series are downloaded individually from TCIA. 40CHECKSUM = None 41 42STRUCTURE_IDS = {"lung_left": 1, "lung_right": 2, "esophagus": 3, "heart": 4, "spinal_cord": 5} 43 44_ROI_NAME_TO_STRUCTURE = { 45 "lung_l": "lung_left", 46 "lung_r": "lung_right", 47 "esophagus": "esophagus", 48 "heart": "heart", 49 "spinalcord": "spinal_cord", 50} 51 52 53class SelectStructures: 54 """Label transform that keeps only the given label ids and sets all other labels to background. 55 56 Args: 57 label_ids: The label ids to keep. 58 """ 59 def __init__(self, label_ids: Sequence[int]): 60 self.label_ids = list(label_ids) 61 62 def __call__(self, labels: np.ndarray) -> np.ndarray: 63 return np.where(np.isin(labels, self.label_ids), labels, 0) 64 65 66def _get_structure_label(roi_number, roi_name): 67 """Map the ROIs of the RTSTRUCT files to the semantic label ids.""" 68 name = roi_name.lower().replace("_", "").replace("-", "").strip() 69 for key, structure in _ROI_NAME_TO_STRUCTURE.items(): 70 if name == key.replace("_", ""): 71 return STRUCTURE_IDS[structure] 72 return None 73 74 75def _get_referenced_series(rtstruct_path): 76 import pydicom 77 78 rtstruct = pydicom.dcmread(rtstruct_path, stop_before_pixels=True) 79 return str( 80 rtstruct.ReferencedFrameOfReferenceSequence[0].RTReferencedStudySequence[0] 81 .RTReferencedSeriesSequence[0].SeriesInstanceUID 82 ) 83 84 85def _preprocess_lctsc(dicom_dir, csv_path, preprocessed_dir): 86 import h5py 87 88 with open(csv_path, "r") as f: 89 rows = list(csv.DictReader(f)) 90 rtstruct_series = {row["Series UID"]: row["Subject ID"] for row in rows if row["Modality"] == "RTSTRUCT"} 91 92 os.makedirs(preprocessed_dir, exist_ok=True) 93 for series_uid, subject_id in tqdm(sorted(rtstruct_series.items()), desc="Preprocess LCTSC"): 94 out_path = os.path.join(preprocessed_dir, f"{subject_id}.h5") 95 if os.path.exists(out_path): 96 continue 97 98 rtstruct_path = glob(os.path.join(dicom_dir, series_uid, "*.dcm"))[0] 99 ct_dir = os.path.join(dicom_dir, _get_referenced_series(rtstruct_path)) 100 volume, geometry = util.load_dicom_series(ct_dir) 101 volume = np.round(volume).astype("int16") 102 labels = util.rasterize_rtstruct(rtstruct_path, geometry, volume.shape, _get_structure_label) 103 104 with h5py.File(out_path, "w") as f: 105 f.create_dataset("raw", data=volume, compression="gzip") 106 f.create_dataset("labels", data=labels, compression="gzip") 107 108 109def get_lctsc_data(path: Union[os.PathLike, str], download: bool = False) -> str: 110 """Download the LCTSC dataset. 111 112 Args: 113 path: Filepath to a folder where the data is downloaded for further processing. 114 download: Whether to download the data if it is not present. 115 116 Returns: 117 Filepath where the preprocessed data is stored. 118 """ 119 # NOTE: The preprocessing below skips volumes that were converted already, so an interrupted run resumes. 120 preprocessed_dir = os.path.join(path, "preprocessed") 121 os.makedirs(path, exist_ok=True) 122 123 # Download the DICOM series (CT and RTSTRUCT) from the TCIA manifest. The series metadata are written 124 # after all series are downloaded, so their presence means the download is complete. 125 dicom_dir = os.path.join(path, "dicom") 126 csv_path = os.path.join(path, "lctsc_series") 127 if not os.path.exists(f"{csv_path}.csv"): 128 util.download_source_tcia( 129 path=os.path.join(path, os.path.basename(URL)), url=URL, dst=dicom_dir, csv_filename=csv_path, 130 download=download, 131 ) 132 133 _preprocess_lctsc(dicom_dir, f"{csv_path}.csv", preprocessed_dir) 134 return preprocessed_dir 135 136 137def get_lctsc_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]: 138 """Get paths to the LCTSC data. 139 140 Args: 141 path: Filepath to a folder where the data is downloaded for further processing. 142 download: Whether to download the data if it is not present. 143 144 Returns: 145 List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels'). 146 """ 147 data_dir = get_lctsc_data(path, download) 148 volume_paths = natsorted(glob(os.path.join(data_dir, "*.h5"))) 149 return volume_paths 150 151 152def get_lctsc_dataset( 153 path: Union[os.PathLike, str], 154 patch_shape: Tuple[int, ...], 155 structures: Optional[Sequence[str]] = None, 156 resize_inputs: bool = False, 157 download: bool = False, 158 **kwargs 159) -> Dataset: 160 """Get the LCTSC dataset for lung and thoracic organ segmentation in CT. 161 162 Args: 163 path: Filepath to a folder where the data is downloaded for further processing. 164 patch_shape: The patch shape to use for training. 165 structures: The structures to use as labels, a subset of 'lung_left', 'lung_right', 'esophagus', 166 'heart' and 'spinal_cord'. All other structures are set to background. By default all 167 structures are used. 168 resize_inputs: Whether to resize inputs to the desired patch shape. 169 download: Whether to download the data if it is not present. 170 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 171 172 Returns: 173 The segmentation dataset. 174 """ 175 volume_paths = get_lctsc_paths(path, download) 176 177 if structures is not None: 178 assert all(structure in STRUCTURE_IDS for structure in structures), f"Invalid structures: {structures}" 179 select_trafo = SelectStructures([STRUCTURE_IDS[structure] for structure in structures]) 180 if "label_transform" in kwargs: 181 kwargs["label_transform"] = Compose(select_trafo, kwargs["label_transform"], is_multi_tensor=False) 182 else: 183 kwargs["label_transform"] = select_trafo 184 185 if resize_inputs: 186 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 187 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 188 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 189 ) 190 191 return torch_em.default_segmentation_dataset( 192 raw_paths=volume_paths, 193 raw_key="raw", 194 label_paths=volume_paths, 195 label_key="labels", 196 patch_shape=patch_shape, 197 is_seg_dataset=True, 198 **kwargs 199 ) 200 201 202def get_lctsc_loader( 203 path: Union[os.PathLike, str], 204 batch_size: int, 205 patch_shape: Tuple[int, ...], 206 structures: Optional[Sequence[str]] = None, 207 resize_inputs: bool = False, 208 download: bool = False, 209 **kwargs 210) -> DataLoader: 211 """Get the LCTSC dataloader for lung and thoracic organ segmentation in CT. 212 213 Args: 214 path: Filepath to a folder where the data is downloaded for further processing. 215 batch_size: The batch size for training. 216 patch_shape: The patch shape to use for training. 217 structures: The structures to use as labels, a subset of 'lung_left', 'lung_right', 'esophagus', 218 'heart' and 'spinal_cord'. All other structures are set to background. By default all 219 structures are used. 220 resize_inputs: Whether to resize inputs to the desired patch shape. 221 download: Whether to download the data if it is not present. 222 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 223 224 Returns: 225 The DataLoader. 226 """ 227 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 228 dataset = get_lctsc_dataset(path, patch_shape, structures, resize_inputs, download, **ds_kwargs) 229 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
54class SelectStructures: 55 """Label transform that keeps only the given label ids and sets all other labels to background. 56 57 Args: 58 label_ids: The label ids to keep. 59 """ 60 def __init__(self, label_ids: Sequence[int]): 61 self.label_ids = list(label_ids) 62 63 def __call__(self, labels: np.ndarray) -> np.ndarray: 64 return np.where(np.isin(labels, self.label_ids), labels, 0)
Label transform that keeps only the given label ids and sets all other labels to background.
Arguments:
- label_ids: The label ids to keep.
110def get_lctsc_data(path: Union[os.PathLike, str], download: bool = False) -> str: 111 """Download the LCTSC dataset. 112 113 Args: 114 path: Filepath to a folder where the data is downloaded for further processing. 115 download: Whether to download the data if it is not present. 116 117 Returns: 118 Filepath where the preprocessed data is stored. 119 """ 120 # NOTE: The preprocessing below skips volumes that were converted already, so an interrupted run resumes. 121 preprocessed_dir = os.path.join(path, "preprocessed") 122 os.makedirs(path, exist_ok=True) 123 124 # Download the DICOM series (CT and RTSTRUCT) from the TCIA manifest. The series metadata are written 125 # after all series are downloaded, so their presence means the download is complete. 126 dicom_dir = os.path.join(path, "dicom") 127 csv_path = os.path.join(path, "lctsc_series") 128 if not os.path.exists(f"{csv_path}.csv"): 129 util.download_source_tcia( 130 path=os.path.join(path, os.path.basename(URL)), url=URL, dst=dicom_dir, csv_filename=csv_path, 131 download=download, 132 ) 133 134 _preprocess_lctsc(dicom_dir, f"{csv_path}.csv", preprocessed_dir) 135 return preprocessed_dir
Download the LCTSC dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the preprocessed data is stored.
138def get_lctsc_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]: 139 """Get paths to the LCTSC data. 140 141 Args: 142 path: Filepath to a folder where the data is downloaded for further processing. 143 download: Whether to download the data if it is not present. 144 145 Returns: 146 List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels'). 147 """ 148 data_dir = get_lctsc_data(path, download) 149 volume_paths = natsorted(glob(os.path.join(data_dir, "*.h5"))) 150 return volume_paths
Get paths to the LCTSC data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').
153def get_lctsc_dataset( 154 path: Union[os.PathLike, str], 155 patch_shape: Tuple[int, ...], 156 structures: Optional[Sequence[str]] = None, 157 resize_inputs: bool = False, 158 download: bool = False, 159 **kwargs 160) -> Dataset: 161 """Get the LCTSC dataset for lung and thoracic organ segmentation in CT. 162 163 Args: 164 path: Filepath to a folder where the data is downloaded for further processing. 165 patch_shape: The patch shape to use for training. 166 structures: The structures to use as labels, a subset of 'lung_left', 'lung_right', 'esophagus', 167 'heart' and 'spinal_cord'. All other structures are set to background. By default all 168 structures are used. 169 resize_inputs: Whether to resize inputs to the desired patch shape. 170 download: Whether to download the data if it is not present. 171 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 172 173 Returns: 174 The segmentation dataset. 175 """ 176 volume_paths = get_lctsc_paths(path, download) 177 178 if structures is not None: 179 assert all(structure in STRUCTURE_IDS for structure in structures), f"Invalid structures: {structures}" 180 select_trafo = SelectStructures([STRUCTURE_IDS[structure] for structure in structures]) 181 if "label_transform" in kwargs: 182 kwargs["label_transform"] = Compose(select_trafo, kwargs["label_transform"], is_multi_tensor=False) 183 else: 184 kwargs["label_transform"] = select_trafo 185 186 if resize_inputs: 187 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 188 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 189 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 190 ) 191 192 return torch_em.default_segmentation_dataset( 193 raw_paths=volume_paths, 194 raw_key="raw", 195 label_paths=volume_paths, 196 label_key="labels", 197 patch_shape=patch_shape, 198 is_seg_dataset=True, 199 **kwargs 200 )
Get the LCTSC dataset for lung and thoracic organ segmentation in CT.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- structures: The structures to use as labels, a subset of 'lung_left', 'lung_right', 'esophagus', 'heart' and 'spinal_cord'. All other structures are set to background. By default all structures are used.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
203def get_lctsc_loader( 204 path: Union[os.PathLike, str], 205 batch_size: int, 206 patch_shape: Tuple[int, ...], 207 structures: Optional[Sequence[str]] = None, 208 resize_inputs: bool = False, 209 download: bool = False, 210 **kwargs 211) -> DataLoader: 212 """Get the LCTSC dataloader for lung and thoracic organ segmentation in CT. 213 214 Args: 215 path: Filepath to a folder where the data is downloaded for further processing. 216 batch_size: The batch size for training. 217 patch_shape: The patch shape to use for training. 218 structures: The structures to use as labels, a subset of 'lung_left', 'lung_right', 'esophagus', 219 'heart' and 'spinal_cord'. All other structures are set to background. By default all 220 structures are used. 221 resize_inputs: Whether to resize inputs to the desired patch shape. 222 download: Whether to download the data if it is not present. 223 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 224 225 Returns: 226 The DataLoader. 227 """ 228 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 229 dataset = get_lctsc_dataset(path, patch_shape, structures, resize_inputs, download, **ds_kwargs) 230 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the LCTSC dataloader for lung and thoracic organ segmentation in CT.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- structures: The structures to use as labels, a subset of 'lung_left', 'lung_right', 'esophagus', 'heart' and 'spinal_cord'. All other structures are set to background. By default all structures are used.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.