torch_em.data.datasets.medical.lctsc

The LCTSC dataset (Lung CT Segmentation Challenge 2017) contains annotations for lung and thoracic organ segmentation in pretreatment CT of non-small cell lung cancer patients.

It consists of 60 CT volumes with manual delineations by an expert, which are distributed as DICOM RTSTRUCT contours. This module rasterizes the contours onto the CT grid (see torch_em.data.datasets.util.rasterize_rtstruct) and stores CT and labels in hdf5 files. The semantic label ids are: 1: left lung ('Lung_L'), 2: right lung ('Lung_R'), 3: esophagus ('Esophagus'), 4: heart ('Heart'), 5: spinal cord ('SpinalCord'). All five structures are annotated for every patient.

NOTE: This requires the pydicom python package.

The dataset is located at https://www.cancerimagingarchive.net/collection/lctsc/.

This dataset is from the publication https://doi.org/10.1002/mp.13141. The data was released at https://doi.org/10.7937/K9/TCIA.2017.3R3FVZ08. Please cite it if you use this dataset in your research.

  1"""The LCTSC dataset (Lung CT Segmentation Challenge 2017) contains annotations for lung and thoracic
  2organ segmentation in pretreatment CT of non-small cell lung cancer patients.
  3
  4It consists of 60 CT volumes with manual delineations by an expert, which are distributed as DICOM
  5RTSTRUCT contours. This module rasterizes the contours onto the CT grid (see
  6`torch_em.data.datasets.util.rasterize_rtstruct`) and stores CT and labels in hdf5 files.
  7The semantic label ids are: 1: left lung ('Lung_L'), 2: right lung ('Lung_R'), 3: esophagus
  8('Esophagus'), 4: heart ('Heart'), 5: spinal cord ('SpinalCord'). All five structures are
  9annotated for every patient.
 10
 11NOTE: This requires the pydicom python package.
 12
 13The dataset is located at https://www.cancerimagingarchive.net/collection/lctsc/.
 14
 15This dataset is from the publication https://doi.org/10.1002/mp.13141.
 16The data was released at https://doi.org/10.7937/K9/TCIA.2017.3R3FVZ08.
 17Please cite it if you use this dataset in your research.
 18"""
 19
 20import os
 21import csv
 22from glob import glob
 23from tqdm import tqdm
 24from natsort import natsorted
 25from typing import Union, Tuple, List, Optional, Sequence
 26
 27import numpy as np
 28
 29from torch.utils.data import Dataset, DataLoader
 30
 31import torch_em
 32from torch_em.transform.generic import Compose
 33
 34from .. import util
 35
 36
 37URL = "https://www.cancerimagingarchive.net/wp-content/uploads/LCTSC_v2_20190508.tcia"
 38
 39# The DICOM series are downloaded individually from TCIA.
 40CHECKSUM = None
 41
 42STRUCTURE_IDS = {"lung_left": 1, "lung_right": 2, "esophagus": 3, "heart": 4, "spinal_cord": 5}
 43
 44_ROI_NAME_TO_STRUCTURE = {
 45    "lung_l": "lung_left",
 46    "lung_r": "lung_right",
 47    "esophagus": "esophagus",
 48    "heart": "heart",
 49    "spinalcord": "spinal_cord",
 50}
 51
 52
 53class SelectStructures:
 54    """Label transform that keeps only the given label ids and sets all other labels to background.
 55
 56    Args:
 57        label_ids: The label ids to keep.
 58    """
 59    def __init__(self, label_ids: Sequence[int]):
 60        self.label_ids = list(label_ids)
 61
 62    def __call__(self, labels: np.ndarray) -> np.ndarray:
 63        return np.where(np.isin(labels, self.label_ids), labels, 0)
 64
 65
 66def _get_structure_label(roi_number, roi_name):
 67    """Map the ROIs of the RTSTRUCT files to the semantic label ids."""
 68    name = roi_name.lower().replace("_", "").replace("-", "").strip()
 69    for key, structure in _ROI_NAME_TO_STRUCTURE.items():
 70        if name == key.replace("_", ""):
 71            return STRUCTURE_IDS[structure]
 72    return None
 73
 74
 75def _get_referenced_series(rtstruct_path):
 76    import pydicom
 77
 78    rtstruct = pydicom.dcmread(rtstruct_path, stop_before_pixels=True)
 79    return str(
 80        rtstruct.ReferencedFrameOfReferenceSequence[0].RTReferencedStudySequence[0]
 81        .RTReferencedSeriesSequence[0].SeriesInstanceUID
 82    )
 83
 84
 85def _preprocess_lctsc(dicom_dir, csv_path, preprocessed_dir):
 86    import h5py
 87
 88    with open(csv_path, "r") as f:
 89        rows = list(csv.DictReader(f))
 90    rtstruct_series = {row["Series UID"]: row["Subject ID"] for row in rows if row["Modality"] == "RTSTRUCT"}
 91
 92    os.makedirs(preprocessed_dir, exist_ok=True)
 93    for series_uid, subject_id in tqdm(sorted(rtstruct_series.items()), desc="Preprocess LCTSC"):
 94        out_path = os.path.join(preprocessed_dir, f"{subject_id}.h5")
 95        if os.path.exists(out_path):
 96            continue
 97
 98        rtstruct_path = glob(os.path.join(dicom_dir, series_uid, "*.dcm"))[0]
 99        ct_dir = os.path.join(dicom_dir, _get_referenced_series(rtstruct_path))
100        volume, geometry = util.load_dicom_series(ct_dir)
101        volume = np.round(volume).astype("int16")
102        labels = util.rasterize_rtstruct(rtstruct_path, geometry, volume.shape, _get_structure_label)
103
104        with h5py.File(out_path, "w") as f:
105            f.create_dataset("raw", data=volume, compression="gzip")
106            f.create_dataset("labels", data=labels, compression="gzip")
107
108
109def get_lctsc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
110    """Download the LCTSC dataset.
111
112    Args:
113        path: Filepath to a folder where the data is downloaded for further processing.
114        download: Whether to download the data if it is not present.
115
116    Returns:
117        Filepath where the preprocessed data is stored.
118    """
119    # NOTE: The preprocessing below skips volumes that were converted already, so an interrupted run resumes.
120    preprocessed_dir = os.path.join(path, "preprocessed")
121    os.makedirs(path, exist_ok=True)
122
123    # Download the DICOM series (CT and RTSTRUCT) from the TCIA manifest. The series metadata are written
124    # after all series are downloaded, so their presence means the download is complete.
125    dicom_dir = os.path.join(path, "dicom")
126    csv_path = os.path.join(path, "lctsc_series")
127    if not os.path.exists(f"{csv_path}.csv"):
128        util.download_source_tcia(
129            path=os.path.join(path, os.path.basename(URL)), url=URL, dst=dicom_dir, csv_filename=csv_path,
130            download=download,
131        )
132
133    _preprocess_lctsc(dicom_dir, f"{csv_path}.csv", preprocessed_dir)
134    return preprocessed_dir
135
136
137def get_lctsc_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
138    """Get paths to the LCTSC data.
139
140    Args:
141        path: Filepath to a folder where the data is downloaded for further processing.
142        download: Whether to download the data if it is not present.
143
144    Returns:
145        List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').
146    """
147    data_dir = get_lctsc_data(path, download)
148    volume_paths = natsorted(glob(os.path.join(data_dir, "*.h5")))
149    return volume_paths
150
151
152def get_lctsc_dataset(
153    path: Union[os.PathLike, str],
154    patch_shape: Tuple[int, ...],
155    structures: Optional[Sequence[str]] = None,
156    resize_inputs: bool = False,
157    download: bool = False,
158    **kwargs
159) -> Dataset:
160    """Get the LCTSC dataset for lung and thoracic organ segmentation in CT.
161
162    Args:
163        path: Filepath to a folder where the data is downloaded for further processing.
164        patch_shape: The patch shape to use for training.
165        structures: The structures to use as labels, a subset of 'lung_left', 'lung_right', 'esophagus',
166            'heart' and 'spinal_cord'. All other structures are set to background. By default all
167            structures are used.
168        resize_inputs: Whether to resize inputs to the desired patch shape.
169        download: Whether to download the data if it is not present.
170        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
171
172    Returns:
173        The segmentation dataset.
174    """
175    volume_paths = get_lctsc_paths(path, download)
176
177    if structures is not None:
178        assert all(structure in STRUCTURE_IDS for structure in structures), f"Invalid structures: {structures}"
179        select_trafo = SelectStructures([STRUCTURE_IDS[structure] for structure in structures])
180        if "label_transform" in kwargs:
181            kwargs["label_transform"] = Compose(select_trafo, kwargs["label_transform"], is_multi_tensor=False)
182        else:
183            kwargs["label_transform"] = select_trafo
184
185    if resize_inputs:
186        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
187        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
188            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
189        )
190
191    return torch_em.default_segmentation_dataset(
192        raw_paths=volume_paths,
193        raw_key="raw",
194        label_paths=volume_paths,
195        label_key="labels",
196        patch_shape=patch_shape,
197        is_seg_dataset=True,
198        **kwargs
199    )
200
201
202def get_lctsc_loader(
203    path: Union[os.PathLike, str],
204    batch_size: int,
205    patch_shape: Tuple[int, ...],
206    structures: Optional[Sequence[str]] = None,
207    resize_inputs: bool = False,
208    download: bool = False,
209    **kwargs
210) -> DataLoader:
211    """Get the LCTSC dataloader for lung and thoracic organ segmentation in CT.
212
213    Args:
214        path: Filepath to a folder where the data is downloaded for further processing.
215        batch_size: The batch size for training.
216        patch_shape: The patch shape to use for training.
217        structures: The structures to use as labels, a subset of 'lung_left', 'lung_right', 'esophagus',
218            'heart' and 'spinal_cord'. All other structures are set to background. By default all
219            structures are used.
220        resize_inputs: Whether to resize inputs to the desired patch shape.
221        download: Whether to download the data if it is not present.
222        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
223
224    Returns:
225        The DataLoader.
226    """
227    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
228    dataset = get_lctsc_dataset(path, patch_shape, structures, resize_inputs, download, **ds_kwargs)
229    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://www.cancerimagingarchive.net/wp-content/uploads/LCTSC_v2_20190508.tcia'
CHECKSUM = None
STRUCTURE_IDS = {'lung_left': 1, 'lung_right': 2, 'esophagus': 3, 'heart': 4, 'spinal_cord': 5}
class SelectStructures:
54class SelectStructures:
55    """Label transform that keeps only the given label ids and sets all other labels to background.
56
57    Args:
58        label_ids: The label ids to keep.
59    """
60    def __init__(self, label_ids: Sequence[int]):
61        self.label_ids = list(label_ids)
62
63    def __call__(self, labels: np.ndarray) -> np.ndarray:
64        return np.where(np.isin(labels, self.label_ids), labels, 0)

Label transform that keeps only the given label ids and sets all other labels to background.

Arguments:
  • label_ids: The label ids to keep.
SelectStructures(label_ids: Sequence[int])
60    def __init__(self, label_ids: Sequence[int]):
61        self.label_ids = list(label_ids)
label_ids
def get_lctsc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
110def get_lctsc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
111    """Download the LCTSC dataset.
112
113    Args:
114        path: Filepath to a folder where the data is downloaded for further processing.
115        download: Whether to download the data if it is not present.
116
117    Returns:
118        Filepath where the preprocessed data is stored.
119    """
120    # NOTE: The preprocessing below skips volumes that were converted already, so an interrupted run resumes.
121    preprocessed_dir = os.path.join(path, "preprocessed")
122    os.makedirs(path, exist_ok=True)
123
124    # Download the DICOM series (CT and RTSTRUCT) from the TCIA manifest. The series metadata are written
125    # after all series are downloaded, so their presence means the download is complete.
126    dicom_dir = os.path.join(path, "dicom")
127    csv_path = os.path.join(path, "lctsc_series")
128    if not os.path.exists(f"{csv_path}.csv"):
129        util.download_source_tcia(
130            path=os.path.join(path, os.path.basename(URL)), url=URL, dst=dicom_dir, csv_filename=csv_path,
131            download=download,
132        )
133
134    _preprocess_lctsc(dicom_dir, f"{csv_path}.csv", preprocessed_dir)
135    return preprocessed_dir

Download the LCTSC dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the preprocessed data is stored.

def get_lctsc_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
138def get_lctsc_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
139    """Get paths to the LCTSC data.
140
141    Args:
142        path: Filepath to a folder where the data is downloaded for further processing.
143        download: Whether to download the data if it is not present.
144
145    Returns:
146        List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').
147    """
148    data_dir = get_lctsc_data(path, download)
149    volume_paths = natsorted(glob(os.path.join(data_dir, "*.h5")))
150    return volume_paths

Get paths to the LCTSC data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').

def get_lctsc_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], structures: Optional[Sequence[str]] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
153def get_lctsc_dataset(
154    path: Union[os.PathLike, str],
155    patch_shape: Tuple[int, ...],
156    structures: Optional[Sequence[str]] = None,
157    resize_inputs: bool = False,
158    download: bool = False,
159    **kwargs
160) -> Dataset:
161    """Get the LCTSC dataset for lung and thoracic organ segmentation in CT.
162
163    Args:
164        path: Filepath to a folder where the data is downloaded for further processing.
165        patch_shape: The patch shape to use for training.
166        structures: The structures to use as labels, a subset of 'lung_left', 'lung_right', 'esophagus',
167            'heart' and 'spinal_cord'. All other structures are set to background. By default all
168            structures are used.
169        resize_inputs: Whether to resize inputs to the desired patch shape.
170        download: Whether to download the data if it is not present.
171        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
172
173    Returns:
174        The segmentation dataset.
175    """
176    volume_paths = get_lctsc_paths(path, download)
177
178    if structures is not None:
179        assert all(structure in STRUCTURE_IDS for structure in structures), f"Invalid structures: {structures}"
180        select_trafo = SelectStructures([STRUCTURE_IDS[structure] for structure in structures])
181        if "label_transform" in kwargs:
182            kwargs["label_transform"] = Compose(select_trafo, kwargs["label_transform"], is_multi_tensor=False)
183        else:
184            kwargs["label_transform"] = select_trafo
185
186    if resize_inputs:
187        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
188        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
189            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
190        )
191
192    return torch_em.default_segmentation_dataset(
193        raw_paths=volume_paths,
194        raw_key="raw",
195        label_paths=volume_paths,
196        label_key="labels",
197        patch_shape=patch_shape,
198        is_seg_dataset=True,
199        **kwargs
200    )

Get the LCTSC dataset for lung and thoracic organ segmentation in CT.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • structures: The structures to use as labels, a subset of 'lung_left', 'lung_right', 'esophagus', 'heart' and 'spinal_cord'. All other structures are set to background. By default all structures are used.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_lctsc_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], structures: Optional[Sequence[str]] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
203def get_lctsc_loader(
204    path: Union[os.PathLike, str],
205    batch_size: int,
206    patch_shape: Tuple[int, ...],
207    structures: Optional[Sequence[str]] = None,
208    resize_inputs: bool = False,
209    download: bool = False,
210    **kwargs
211) -> DataLoader:
212    """Get the LCTSC dataloader for lung and thoracic organ segmentation in CT.
213
214    Args:
215        path: Filepath to a folder where the data is downloaded for further processing.
216        batch_size: The batch size for training.
217        patch_shape: The patch shape to use for training.
218        structures: The structures to use as labels, a subset of 'lung_left', 'lung_right', 'esophagus',
219            'heart' and 'spinal_cord'. All other structures are set to background. By default all
220            structures are used.
221        resize_inputs: Whether to resize inputs to the desired patch shape.
222        download: Whether to download the data if it is not present.
223        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
224
225    Returns:
226        The DataLoader.
227    """
228    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
229    dataset = get_lctsc_dataset(path, patch_shape, structures, resize_inputs, download, **ds_kwargs)
230    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the LCTSC dataloader for lung and thoracic organ segmentation in CT.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • structures: The structures to use as labels, a subset of 'lung_left', 'lung_right', 'esophagus', 'heart' and 'spinal_cord'. All other structures are set to background. By default all structures are used.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.