torch_em.data.datasets.medical.pancreatic_ct_cbct_seg

The Pancreatic-CT-CBCT-SEG dataset contains annotations for organs at risk in planning CT of pancreatic cancer patients undergoing ablative radiation therapy.

It consists of 40 patients, each with a breath-hold planning CT and two cone-beam CTs (CBCTs) from their treatment. This module only uses the planning CT, which is distributed as a DICOM series with a DICOM RTSTRUCT contour of the organs at risk. The contours are rasterized onto the CT grid (see torch_em.data.datasets.util.rasterize_rtstruct) and CT and labels are stored in hdf5 files. The semantic label ids are: 1: small bowel, 2: stomach and duodenum. The RTSTRUCT files also contain lung and volume-of-interest contours that are not used here, since they are not organ-at-risk targets.

NOTE: This requires the pydicom python package.

The dataset is located at https://www.cancerimagingarchive.net/collection/pancreatic-ct-cbct-seg/.

This dataset is from the publication https://doi.org/10.1038/s41597-022-01758-9. The data was released at https://doi.org/10.7937/TCIA.ESHQ-4D90. Please cite it if you use this dataset in your research.

  1"""The Pancreatic-CT-CBCT-SEG dataset contains annotations for organs at risk in planning CT of
  2pancreatic cancer patients undergoing ablative radiation therapy.
  3
  4It consists of 40 patients, each with a breath-hold planning CT and two cone-beam CTs (CBCTs) from
  5their treatment. This module only uses the planning CT, which is distributed as a DICOM series with a
  6DICOM RTSTRUCT contour of the organs at risk. The contours are rasterized onto the CT grid (see
  7`torch_em.data.datasets.util.rasterize_rtstruct`) and CT and labels are stored in hdf5 files.
  8The semantic label ids are: 1: small bowel, 2: stomach and duodenum. The RTSTRUCT files also contain
  9lung and volume-of-interest contours that are not used here, since they are not organ-at-risk targets.
 10
 11NOTE: This requires the pydicom python package.
 12
 13The dataset is located at https://www.cancerimagingarchive.net/collection/pancreatic-ct-cbct-seg/.
 14
 15This dataset is from the publication https://doi.org/10.1038/s41597-022-01758-9.
 16The data was released at https://doi.org/10.7937/TCIA.ESHQ-4D90.
 17Please cite it if you use this dataset in your research.
 18"""
 19
 20import os
 21import csv
 22from glob import glob
 23from tqdm import tqdm
 24from natsort import natsorted
 25from typing import Union, Tuple, List, Optional, Sequence
 26
 27import numpy as np
 28
 29from torch.utils.data import Dataset, DataLoader
 30
 31import torch_em
 32from torch_em.transform.generic import Compose
 33
 34from .. import util
 35
 36
 37URL = "https://www.cancerimagingarchive.net/wp-content/uploads/Pancreatic-CT-CBCT-SEG_v2_20220823.tcia"
 38
 39# The DICOM series are downloaded individually from TCIA.
 40CHECKSUM = None
 41
 42STRUCTURE_IDS = {"bowel": 1, "stomach_duodenum": 2}
 43
 44
 45class SelectStructures:
 46    """Label transform that keeps only the given label ids and sets all other labels to background.
 47
 48    Args:
 49        label_ids: The label ids to keep.
 50    """
 51    def __init__(self, label_ids: Sequence[int]):
 52        self.label_ids = list(label_ids)
 53
 54    def __call__(self, labels: np.ndarray) -> np.ndarray:
 55        return np.where(np.isin(labels, self.label_ids), labels, 0)
 56
 57
 58def _get_structure_label(roi_number, roi_name):
 59    """Map the ROIs of the planning CT RTSTRUCT files to the semantic label ids.
 60
 61    Only the small bowel and stomach/duodenum organ-at-risk contours are used; lung and
 62    volume-of-interest contours (present in the same file) are ignored.
 63    """
 64    name = roi_name.lower()
 65    if "bowel" in name:
 66        return STRUCTURE_IDS["bowel"]
 67    if "stomach" in name or "duo" in name:
 68        return STRUCTURE_IDS["stomach_duodenum"]
 69    return None
 70
 71
 72def _is_planning_ct_rtstruct(rtstruct_path):
 73    """Check whether an RTSTRUCT file belongs to the planning CT, based on its ROI names.
 74
 75    The planning CT contours are the only ones with a '_planCT' suffix (e.g. 'Bowel_sm_planCT'),
 76    which distinguishes them from the CBCT contours in the same collection.
 77    """
 78    import pydicom
 79
 80    rtstruct = pydicom.dcmread(rtstruct_path, stop_before_pixels=True)
 81    roi_names = [str(roi.ROIName).lower() for roi in rtstruct.StructureSetROISequence]
 82    return any("planct" in name.replace("_", "") for name in roi_names)
 83
 84
 85def _get_referenced_series(rtstruct_path):
 86    import pydicom
 87
 88    rtstruct = pydicom.dcmread(rtstruct_path, stop_before_pixels=True)
 89    return str(
 90        rtstruct.ReferencedFrameOfReferenceSequence[0].RTReferencedStudySequence[0]
 91        .RTReferencedSeriesSequence[0].SeriesInstanceUID
 92    )
 93
 94
 95def _preprocess_pancreatic_ct_cbct_seg(dicom_dir, csv_path, preprocessed_dir):
 96    import h5py
 97
 98    with open(csv_path, "r") as f:
 99        rows = list(csv.DictReader(f))
100    rtstruct_series = {row["Series UID"]: row["Subject ID"] for row in rows if row["Modality"] == "RTSTRUCT"}
101
102    os.makedirs(preprocessed_dir, exist_ok=True)
103    for series_uid, subject_id in tqdm(sorted(rtstruct_series.items()), desc="Preprocess Pancreatic-CT-CBCT-SEG"):
104        out_path = os.path.join(preprocessed_dir, f"{subject_id}.h5")
105        if os.path.exists(out_path):
106            continue
107
108        rtstruct_path = glob(os.path.join(dicom_dir, series_uid, "*.dcm"))[0]
109        if not _is_planning_ct_rtstruct(rtstruct_path):
110            continue  # This is a CBCT RTSTRUCT, which is not used by this module.
111
112        ct_dir = os.path.join(dicom_dir, _get_referenced_series(rtstruct_path))
113        volume, geometry = util.load_dicom_series(ct_dir)
114        volume = np.round(volume).astype("int16")
115        labels = util.rasterize_rtstruct(rtstruct_path, geometry, volume.shape, _get_structure_label)
116
117        # Written to a temporary path and renamed atomically, so that a run interrupted mid-write
118        # never leaves a stale, incomplete file at 'out_path' for a later run to mistake as done.
119        tmp_path = f"{out_path}.tmp"
120        with h5py.File(tmp_path, "w") as f:
121            f.create_dataset("raw", data=volume, compression="gzip")
122            f.create_dataset("labels", data=labels, compression="gzip")
123        os.replace(tmp_path, out_path)
124
125
126def get_pancreatic_ct_cbct_seg_data(path: Union[os.PathLike, str], download: bool = False) -> str:
127    """Download the Pancreatic-CT-CBCT-SEG dataset.
128
129    Args:
130        path: Filepath to a folder where the data is downloaded for further processing.
131        download: Whether to download the data if it is not present.
132
133    Returns:
134        Filepath where the preprocessed data is stored.
135    """
136    # NOTE: The preprocessing below skips volumes that were converted already, so an interrupted run resumes.
137    preprocessed_dir = os.path.join(path, "preprocessed")
138    os.makedirs(path, exist_ok=True)
139
140    # Download the full collection (CT, CBCT and RTSTRUCT series) from the TCIA manifest. Only the
141    # planning CT and its RTSTRUCT are used for preprocessing, see `_is_planning_ct_rtstruct`.
142    dicom_dir = os.path.join(path, "dicom")
143    csv_path = os.path.join(path, "pancreatic_ct_cbct_seg_series")
144    if not os.path.exists(f"{csv_path}.csv"):
145        util.download_source_tcia(
146            path=os.path.join(path, os.path.basename(URL)), url=URL, dst=dicom_dir, csv_filename=csv_path,
147            download=download,
148        )
149
150    _preprocess_pancreatic_ct_cbct_seg(dicom_dir, f"{csv_path}.csv", preprocessed_dir)
151    return preprocessed_dir
152
153
154def get_pancreatic_ct_cbct_seg_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
155    """Get paths to the Pancreatic-CT-CBCT-SEG data.
156
157    Args:
158        path: Filepath to a folder where the data is downloaded for further processing.
159        download: Whether to download the data if it is not present.
160
161    Returns:
162        List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').
163    """
164    data_dir = get_pancreatic_ct_cbct_seg_data(path, download)
165    volume_paths = natsorted(glob(os.path.join(data_dir, "*.h5")))
166    return volume_paths
167
168
169def get_pancreatic_ct_cbct_seg_dataset(
170    path: Union[os.PathLike, str],
171    patch_shape: Tuple[int, ...],
172    structures: Optional[Sequence[str]] = None,
173    resize_inputs: bool = False,
174    download: bool = False,
175    **kwargs
176) -> Dataset:
177    """Get the Pancreatic-CT-CBCT-SEG dataset for organ-at-risk segmentation in planning CT.
178
179    Args:
180        path: Filepath to a folder where the data is downloaded for further processing.
181        patch_shape: The patch shape to use for training.
182        structures: The structures to use as labels, a subset of 'bowel' and 'stomach_duodenum'.
183            All other structures are set to background. By default both structures are used.
184        resize_inputs: Whether to resize inputs to the desired patch shape.
185        download: Whether to download the data if it is not present.
186        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
187
188    Returns:
189        The segmentation dataset.
190    """
191    volume_paths = get_pancreatic_ct_cbct_seg_paths(path, download)
192
193    if structures is not None:
194        assert all(structure in STRUCTURE_IDS for structure in structures), f"Invalid structures: {structures}"
195        select_trafo = SelectStructures([STRUCTURE_IDS[structure] for structure in structures])
196        if "label_transform" in kwargs:
197            kwargs["label_transform"] = Compose(select_trafo, kwargs["label_transform"], is_multi_tensor=False)
198        else:
199            kwargs["label_transform"] = select_trafo
200
201    if resize_inputs:
202        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
203        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
204            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
205        )
206
207    return torch_em.default_segmentation_dataset(
208        raw_paths=volume_paths,
209        raw_key="raw",
210        label_paths=volume_paths,
211        label_key="labels",
212        patch_shape=patch_shape,
213        is_seg_dataset=True,
214        **kwargs
215    )
216
217
218def get_pancreatic_ct_cbct_seg_loader(
219    path: Union[os.PathLike, str],
220    batch_size: int,
221    patch_shape: Tuple[int, ...],
222    structures: Optional[Sequence[str]] = None,
223    resize_inputs: bool = False,
224    download: bool = False,
225    **kwargs
226) -> DataLoader:
227    """Get the Pancreatic-CT-CBCT-SEG dataloader for organ-at-risk segmentation in planning CT.
228
229    Args:
230        path: Filepath to a folder where the data is downloaded for further processing.
231        batch_size: The batch size for training.
232        patch_shape: The patch shape to use for training.
233        structures: The structures to use as labels, a subset of 'bowel' and 'stomach_duodenum'.
234            All other structures are set to background. By default both structures are used.
235        resize_inputs: Whether to resize inputs to the desired patch shape.
236        download: Whether to download the data if it is not present.
237        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
238
239    Returns:
240        The DataLoader.
241    """
242    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
243    dataset = get_pancreatic_ct_cbct_seg_dataset(path, patch_shape, structures, resize_inputs, download, **ds_kwargs)
244    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://www.cancerimagingarchive.net/wp-content/uploads/Pancreatic-CT-CBCT-SEG_v2_20220823.tcia'
CHECKSUM = None
STRUCTURE_IDS = {'bowel': 1, 'stomach_duodenum': 2}
class SelectStructures:
46class SelectStructures:
47    """Label transform that keeps only the given label ids and sets all other labels to background.
48
49    Args:
50        label_ids: The label ids to keep.
51    """
52    def __init__(self, label_ids: Sequence[int]):
53        self.label_ids = list(label_ids)
54
55    def __call__(self, labels: np.ndarray) -> np.ndarray:
56        return np.where(np.isin(labels, self.label_ids), labels, 0)

Label transform that keeps only the given label ids and sets all other labels to background.

Arguments:
  • label_ids: The label ids to keep.
SelectStructures(label_ids: Sequence[int])
52    def __init__(self, label_ids: Sequence[int]):
53        self.label_ids = list(label_ids)
label_ids
def get_pancreatic_ct_cbct_seg_data(path: Union[os.PathLike, str], download: bool = False) -> str:
127def get_pancreatic_ct_cbct_seg_data(path: Union[os.PathLike, str], download: bool = False) -> str:
128    """Download the Pancreatic-CT-CBCT-SEG dataset.
129
130    Args:
131        path: Filepath to a folder where the data is downloaded for further processing.
132        download: Whether to download the data if it is not present.
133
134    Returns:
135        Filepath where the preprocessed data is stored.
136    """
137    # NOTE: The preprocessing below skips volumes that were converted already, so an interrupted run resumes.
138    preprocessed_dir = os.path.join(path, "preprocessed")
139    os.makedirs(path, exist_ok=True)
140
141    # Download the full collection (CT, CBCT and RTSTRUCT series) from the TCIA manifest. Only the
142    # planning CT and its RTSTRUCT are used for preprocessing, see `_is_planning_ct_rtstruct`.
143    dicom_dir = os.path.join(path, "dicom")
144    csv_path = os.path.join(path, "pancreatic_ct_cbct_seg_series")
145    if not os.path.exists(f"{csv_path}.csv"):
146        util.download_source_tcia(
147            path=os.path.join(path, os.path.basename(URL)), url=URL, dst=dicom_dir, csv_filename=csv_path,
148            download=download,
149        )
150
151    _preprocess_pancreatic_ct_cbct_seg(dicom_dir, f"{csv_path}.csv", preprocessed_dir)
152    return preprocessed_dir

Download the Pancreatic-CT-CBCT-SEG dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the preprocessed data is stored.

def get_pancreatic_ct_cbct_seg_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
155def get_pancreatic_ct_cbct_seg_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
156    """Get paths to the Pancreatic-CT-CBCT-SEG data.
157
158    Args:
159        path: Filepath to a folder where the data is downloaded for further processing.
160        download: Whether to download the data if it is not present.
161
162    Returns:
163        List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').
164    """
165    data_dir = get_pancreatic_ct_cbct_seg_data(path, download)
166    volume_paths = natsorted(glob(os.path.join(data_dir, "*.h5")))
167    return volume_paths

Get paths to the Pancreatic-CT-CBCT-SEG data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').

def get_pancreatic_ct_cbct_seg_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], structures: Optional[Sequence[str]] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
170def get_pancreatic_ct_cbct_seg_dataset(
171    path: Union[os.PathLike, str],
172    patch_shape: Tuple[int, ...],
173    structures: Optional[Sequence[str]] = None,
174    resize_inputs: bool = False,
175    download: bool = False,
176    **kwargs
177) -> Dataset:
178    """Get the Pancreatic-CT-CBCT-SEG dataset for organ-at-risk segmentation in planning CT.
179
180    Args:
181        path: Filepath to a folder where the data is downloaded for further processing.
182        patch_shape: The patch shape to use for training.
183        structures: The structures to use as labels, a subset of 'bowel' and 'stomach_duodenum'.
184            All other structures are set to background. By default both structures are used.
185        resize_inputs: Whether to resize inputs to the desired patch shape.
186        download: Whether to download the data if it is not present.
187        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
188
189    Returns:
190        The segmentation dataset.
191    """
192    volume_paths = get_pancreatic_ct_cbct_seg_paths(path, download)
193
194    if structures is not None:
195        assert all(structure in STRUCTURE_IDS for structure in structures), f"Invalid structures: {structures}"
196        select_trafo = SelectStructures([STRUCTURE_IDS[structure] for structure in structures])
197        if "label_transform" in kwargs:
198            kwargs["label_transform"] = Compose(select_trafo, kwargs["label_transform"], is_multi_tensor=False)
199        else:
200            kwargs["label_transform"] = select_trafo
201
202    if resize_inputs:
203        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
204        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
205            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
206        )
207
208    return torch_em.default_segmentation_dataset(
209        raw_paths=volume_paths,
210        raw_key="raw",
211        label_paths=volume_paths,
212        label_key="labels",
213        patch_shape=patch_shape,
214        is_seg_dataset=True,
215        **kwargs
216    )

Get the Pancreatic-CT-CBCT-SEG dataset for organ-at-risk segmentation in planning CT.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • structures: The structures to use as labels, a subset of 'bowel' and 'stomach_duodenum'. All other structures are set to background. By default both structures are used.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_pancreatic_ct_cbct_seg_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], structures: Optional[Sequence[str]] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
219def get_pancreatic_ct_cbct_seg_loader(
220    path: Union[os.PathLike, str],
221    batch_size: int,
222    patch_shape: Tuple[int, ...],
223    structures: Optional[Sequence[str]] = None,
224    resize_inputs: bool = False,
225    download: bool = False,
226    **kwargs
227) -> DataLoader:
228    """Get the Pancreatic-CT-CBCT-SEG dataloader for organ-at-risk segmentation in planning CT.
229
230    Args:
231        path: Filepath to a folder where the data is downloaded for further processing.
232        batch_size: The batch size for training.
233        patch_shape: The patch shape to use for training.
234        structures: The structures to use as labels, a subset of 'bowel' and 'stomach_duodenum'.
235            All other structures are set to background. By default both structures are used.
236        resize_inputs: Whether to resize inputs to the desired patch shape.
237        download: Whether to download the data if it is not present.
238        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
239
240    Returns:
241        The DataLoader.
242    """
243    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
244    dataset = get_pancreatic_ct_cbct_seg_dataset(path, patch_shape, structures, resize_inputs, download, **ds_kwargs)
245    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the Pancreatic-CT-CBCT-SEG dataloader for organ-at-risk segmentation in planning CT.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • structures: The structures to use as labels, a subset of 'bowel' and 'stomach_duodenum'. All other structures are set to background. By default both structures are used.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.