torch_em.data.datasets.medical.qin_lungct_seg

The QIN-LungCT-Seg dataset contains annotations for lung tumors in CT.

It consists of repeated segmentations of the same tumors, drawn either manually or by one of several semi-automated algorithms, across four source collections (LIDC-IDRI, RIDER Lung CT, QIN LUNG CT and a CT lung phantom). Each segmentation is a separate DICOM-SEG object paired here with the exact CT series it references, so the same tumor can appear multiple times with different segmentations - useful for studying inter-algorithm and inter-rater variability, not just as extra training pairs.

NOTE: This requires the pydicom python package.

The dataset is located at https://www.cancerimagingarchive.net/collection/qin-lungct-seg/.

The data was released at https://doi.org/10.7937/K9/TCIA.2015.1BUVFJR7. Please cite it if you use this dataset in your research.

  1"""The QIN-LungCT-Seg dataset contains annotations for lung tumors in CT.
  2
  3It consists of repeated segmentations of the same tumors, drawn either manually or by one of several
  4semi-automated algorithms, across four source collections (LIDC-IDRI, RIDER Lung CT, QIN LUNG CT and a
  5CT lung phantom). Each segmentation is a separate DICOM-SEG object paired here with the exact CT
  6series it references, so the same tumor can appear multiple times with different segmentations - useful
  7for studying inter-algorithm and inter-rater variability, not just as extra training pairs.
  8
  9NOTE: This requires the pydicom python package.
 10
 11The dataset is located at https://www.cancerimagingarchive.net/collection/qin-lungct-seg/.
 12
 13The data was released at https://doi.org/10.7937/K9/TCIA.2015.1BUVFJR7.
 14Please cite it if you use this dataset in your research.
 15"""
 16
 17import os
 18import csv
 19from glob import glob
 20from tqdm import tqdm
 21from natsort import natsorted
 22from typing import Union, Tuple, List
 23
 24from torch.utils.data import Dataset, DataLoader
 25
 26import torch_em
 27
 28from .adrenal_acc import _load_dicom_volume, _load_dicom_seg, _resample_labels
 29from .. import util
 30
 31
 32URL = "https://www.cancerimagingarchive.net/wp-content/uploads/QIN-Multi-site-Lung-CTs-and-SEG-minus-Stanford.tcia"  # noqa
 33
 34# The DICOM series are downloaded individually from TCIA.
 35CHECKSUM = None
 36
 37
 38def _get_referenced_series(seg_path):
 39    import pydicom
 40
 41    seg = pydicom.dcmread(seg_path, stop_before_pixels=True)
 42    return str(seg.ReferencedSeriesSequence[0].SeriesInstanceUID)
 43
 44
 45def _preprocess_qin_lungct_seg(dicom_dir, csv_path, preprocessed_dir):
 46    import h5py
 47
 48    with open(csv_path, "r") as f:
 49        rows = list(csv.DictReader(f))
 50    seg_series = [row["Series UID"] for row in rows if row["Modality"] == "SEG"]
 51
 52    os.makedirs(preprocessed_dir, exist_ok=True)
 53    for series_uid in tqdm(sorted(seg_series), desc="Preprocess QIN-LungCT-Seg"):
 54        out_path = os.path.join(preprocessed_dir, f"{series_uid}.h5")
 55        if os.path.exists(out_path):
 56            continue
 57
 58        seg_paths = glob(os.path.join(dicom_dir, series_uid, "*.dcm"))
 59        if not seg_paths:
 60            continue
 61
 62        ct_dir = os.path.join(dicom_dir, _get_referenced_series(seg_paths[0]))
 63        if not glob(os.path.join(ct_dir, "*.dcm")):
 64            continue
 65
 66        volume, ct_affine = _load_dicom_volume(ct_dir)
 67        seg_labels, seg_affine = _load_dicom_seg(seg_paths[0])
 68        labels = _resample_labels(seg_labels, seg_affine, volume.shape, ct_affine)
 69        if labels.max() == 0:
 70            continue
 71
 72        with h5py.File(out_path, "w") as f:
 73            f.create_dataset("raw", data=volume, compression="gzip")
 74            f.create_dataset("labels", data=labels, compression="gzip")
 75
 76
 77def get_qin_lungct_seg_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 78    """Download the QIN-LungCT-Seg dataset.
 79
 80    Args:
 81        path: Filepath to a folder where the data is downloaded for further processing.
 82        download: Whether to download the data if it is not present.
 83
 84    Returns:
 85        Filepath where the preprocessed data is stored.
 86    """
 87    # NOTE: The preprocessing below skips volumes that were converted already, so an interrupted run resumes.
 88    preprocessed_dir = os.path.join(path, "preprocessed")
 89    os.makedirs(path, exist_ok=True)
 90
 91    dicom_dir = os.path.join(path, "dicom")
 92    csv_path = os.path.join(path, "qin_lungct_seg_series")
 93    if not os.path.exists(f"{csv_path}.csv"):
 94        util.download_source_tcia(
 95            path=os.path.join(path, os.path.basename(URL)), url=URL, dst=dicom_dir, csv_filename=csv_path,
 96            download=download,
 97        )
 98
 99    _preprocess_qin_lungct_seg(dicom_dir, f"{csv_path}.csv", preprocessed_dir)
100    return preprocessed_dir
101
102
103def get_qin_lungct_seg_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
104    """Get paths to the QIN-LungCT-Seg data.
105
106    Args:
107        path: Filepath to a folder where the data is downloaded for further processing.
108        download: Whether to download the data if it is not present.
109
110    Returns:
111        List of filepaths for the stored data.
112    """
113    preprocessed_dir = get_qin_lungct_seg_data(path, download)
114    volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5")))
115    assert len(volume_paths) > 0, f"Could not find any preprocessed volumes in '{preprocessed_dir}'."
116    return volume_paths
117
118
119def get_qin_lungct_seg_dataset(
120    path: Union[os.PathLike, str],
121    patch_shape: Tuple[int, ...],
122    resize_inputs: bool = False,
123    download: bool = False,
124    **kwargs
125) -> Dataset:
126    """Get the QIN-LungCT-Seg dataset for lung tumor segmentation.
127
128    Args:
129        path: Filepath to a folder where the data is downloaded for further processing.
130        patch_shape: The patch shape to use for training.
131        resize_inputs: Whether to resize inputs to the desired patch shape.
132        download: Whether to download the data if it is not present.
133        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
134
135    Returns:
136        The segmentation dataset.
137    """
138    volume_paths = get_qin_lungct_seg_paths(path, download)
139
140    if resize_inputs:
141        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
142        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
143            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
144        )
145
146    return torch_em.default_segmentation_dataset(
147        raw_paths=volume_paths,
148        raw_key="raw",
149        label_paths=volume_paths,
150        label_key="labels",
151        patch_shape=patch_shape,
152        is_seg_dataset=True,
153        **kwargs
154    )
155
156
157def get_qin_lungct_seg_loader(
158    path: Union[os.PathLike, str],
159    batch_size: int,
160    patch_shape: Tuple[int, ...],
161    resize_inputs: bool = False,
162    download: bool = False,
163    **kwargs
164) -> DataLoader:
165    """Get the QIN-LungCT-Seg dataloader for lung tumor segmentation.
166
167    Args:
168        path: Filepath to a folder where the data is downloaded for further processing.
169        batch_size: The batch size for training.
170        patch_shape: The patch shape to use for training.
171        resize_inputs: Whether to resize inputs to the desired patch shape.
172        download: Whether to download the data if it is not present.
173        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
174
175    Returns:
176        The DataLoader.
177    """
178    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
179    dataset = get_qin_lungct_seg_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
180    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://www.cancerimagingarchive.net/wp-content/uploads/QIN-Multi-site-Lung-CTs-and-SEG-minus-Stanford.tcia'
CHECKSUM = None
def get_qin_lungct_seg_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 78def get_qin_lungct_seg_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 79    """Download the QIN-LungCT-Seg dataset.
 80
 81    Args:
 82        path: Filepath to a folder where the data is downloaded for further processing.
 83        download: Whether to download the data if it is not present.
 84
 85    Returns:
 86        Filepath where the preprocessed data is stored.
 87    """
 88    # NOTE: The preprocessing below skips volumes that were converted already, so an interrupted run resumes.
 89    preprocessed_dir = os.path.join(path, "preprocessed")
 90    os.makedirs(path, exist_ok=True)
 91
 92    dicom_dir = os.path.join(path, "dicom")
 93    csv_path = os.path.join(path, "qin_lungct_seg_series")
 94    if not os.path.exists(f"{csv_path}.csv"):
 95        util.download_source_tcia(
 96            path=os.path.join(path, os.path.basename(URL)), url=URL, dst=dicom_dir, csv_filename=csv_path,
 97            download=download,
 98        )
 99
100    _preprocess_qin_lungct_seg(dicom_dir, f"{csv_path}.csv", preprocessed_dir)
101    return preprocessed_dir

Download the QIN-LungCT-Seg dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the preprocessed data is stored.

def get_qin_lungct_seg_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
104def get_qin_lungct_seg_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
105    """Get paths to the QIN-LungCT-Seg data.
106
107    Args:
108        path: Filepath to a folder where the data is downloaded for further processing.
109        download: Whether to download the data if it is not present.
110
111    Returns:
112        List of filepaths for the stored data.
113    """
114    preprocessed_dir = get_qin_lungct_seg_data(path, download)
115    volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5")))
116    assert len(volume_paths) > 0, f"Could not find any preprocessed volumes in '{preprocessed_dir}'."
117    return volume_paths

Get paths to the QIN-LungCT-Seg data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the stored data.

def get_qin_lungct_seg_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
120def get_qin_lungct_seg_dataset(
121    path: Union[os.PathLike, str],
122    patch_shape: Tuple[int, ...],
123    resize_inputs: bool = False,
124    download: bool = False,
125    **kwargs
126) -> Dataset:
127    """Get the QIN-LungCT-Seg dataset for lung tumor segmentation.
128
129    Args:
130        path: Filepath to a folder where the data is downloaded for further processing.
131        patch_shape: The patch shape to use for training.
132        resize_inputs: Whether to resize inputs to the desired patch shape.
133        download: Whether to download the data if it is not present.
134        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
135
136    Returns:
137        The segmentation dataset.
138    """
139    volume_paths = get_qin_lungct_seg_paths(path, download)
140
141    if resize_inputs:
142        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
143        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
144            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
145        )
146
147    return torch_em.default_segmentation_dataset(
148        raw_paths=volume_paths,
149        raw_key="raw",
150        label_paths=volume_paths,
151        label_key="labels",
152        patch_shape=patch_shape,
153        is_seg_dataset=True,
154        **kwargs
155    )

Get the QIN-LungCT-Seg dataset for lung tumor segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_qin_lungct_seg_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
158def get_qin_lungct_seg_loader(
159    path: Union[os.PathLike, str],
160    batch_size: int,
161    patch_shape: Tuple[int, ...],
162    resize_inputs: bool = False,
163    download: bool = False,
164    **kwargs
165) -> DataLoader:
166    """Get the QIN-LungCT-Seg dataloader for lung tumor segmentation.
167
168    Args:
169        path: Filepath to a folder where the data is downloaded for further processing.
170        batch_size: The batch size for training.
171        patch_shape: The patch shape to use for training.
172        resize_inputs: Whether to resize inputs to the desired patch shape.
173        download: Whether to download the data if it is not present.
174        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
175
176    Returns:
177        The DataLoader.
178    """
179    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
180    dataset = get_qin_lungct_seg_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
181    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the QIN-LungCT-Seg dataloader for lung tumor segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.