torch_em.data.datasets.medical.adrenal_acc

The Adrenal-ACC-Ki67-Seg dataset contains annotations for adrenocortical carcinoma segmentation in contrast-enhanced abdominal CT.

It consists of 53 CT volumes (one segmented CT series per patient, the collection also contains 71 additional unsegmented CT series of the same patients) with binary tumor labels (1: adrenal tumor). The CT scans are distributed as DICOM series and the labels as DICOM-SEG objects, which are converted and stored in hdf5 files by this module.

NOTE: This requires the pydicom python package.

The dataset is located at https://www.cancerimagingarchive.net/collection/adrenal-acc-ki67-seg/.

This dataset is from the publication https://doi.org/10.1016/j.crad.2020.01.012. The data was released at https://doi.org/10.7937/1FPG-VM46. Please cite it if you use this dataset in your research.

  1"""The Adrenal-ACC-Ki67-Seg dataset contains annotations for adrenocortical carcinoma segmentation
  2in contrast-enhanced abdominal CT.
  3
  4It consists of 53 CT volumes (one segmented CT series per patient, the collection also contains 71 additional
  5unsegmented CT series of the same patients) with binary tumor labels (1: adrenal tumor). The CT scans are
  6distributed as DICOM series and the labels as DICOM-SEG objects, which are converted and stored in hdf5 files
  7by this module.
  8
  9NOTE: This requires the pydicom python package.
 10
 11The dataset is located at https://www.cancerimagingarchive.net/collection/adrenal-acc-ki67-seg/.
 12
 13This dataset is from the publication https://doi.org/10.1016/j.crad.2020.01.012.
 14The data was released at https://doi.org/10.7937/1FPG-VM46.
 15Please cite it if you use this dataset in your research.
 16"""
 17
 18import os
 19import csv
 20from glob import glob
 21from tqdm import tqdm
 22from natsort import natsorted
 23from typing import Union, Tuple, List
 24
 25import numpy as np
 26
 27from torch.utils.data import Dataset, DataLoader
 28
 29import torch_em
 30
 31from .. import util
 32
 33
 34URL = "https://www.cancerimagingarchive.net/wp-content/uploads/Adrenal-ACC-Ki67-Seg_v1.tcia"
 35
 36# The DICOM series are downloaded individually from TCIA.
 37CHECKSUM = None
 38
 39LABEL_IDS = {"tumor": 1}
 40
 41
 42def _load_dicom_volume(series_dir):
 43    """Stack a DICOM series into a volume with axes (z, y, x) and slices sorted along the slice normal.
 44
 45    Applies the rescale slope/intercept if present (e.g. to get Hounsfield units for CT); MR series
 46    usually do not carry these tags, in which case the stored pixel values are used as-is, per the
 47    DICOM standard's own default of a slope of 1 and an intercept of 0.
 48
 49    Returns the volume and the affine matrix that maps voxel indices (z, y, x) to DICOM patient
 50    coordinates.
 51    """
 52    import pydicom
 53
 54    slices = [pydicom.dcmread(dcm_path) for dcm_path in natsorted(glob(os.path.join(series_dir, "*.dcm")))]
 55    orientation = np.array([float(v) for v in slices[0].ImageOrientationPatient])
 56    row_dir, col_dir = orientation[:3], orientation[3:]
 57    normal = np.cross(row_dir, col_dir)
 58    slices.sort(key=lambda dcm: np.dot([float(v) for v in dcm.ImagePositionPatient], normal))
 59
 60    volume = np.stack([dcm.pixel_array for dcm in slices]).astype("float32")
 61    volume = volume * float(getattr(slices[0], "RescaleSlope", 1.0)) + float(getattr(slices[0], "RescaleIntercept", 0.0))  # noqa
 62    volume = np.round(volume).astype("int16")
 63
 64    positions = np.array([[float(v) for v in dcm.ImagePositionPatient] for dcm in slices])
 65    spacing = [float(v) for v in slices[0].PixelSpacing]  # The spacing between rows and between columns.
 66    affine = np.eye(4)
 67    affine[:3, 0] = (positions[-1] - positions[0]) / (len(slices) - 1)
 68    affine[:3, 1] = col_dir * spacing[0]
 69    affine[:3, 2] = row_dir * spacing[1]
 70    affine[:3, 3] = positions[0]
 71    return volume, affine
 72
 73
 74def _load_dicom_seg(seg_path):
 75    """Load a DICOM-SEG object as a label volume with axes (z, y, x), where the segment number is used as label id.
 76
 77    Returns the label volume and the affine matrix that maps its voxel indices to DICOM patient coordinates.
 78    """
 79    import pydicom
 80
 81    seg = pydicom.dcmread(seg_path)
 82    frames = seg.pixel_array
 83    if frames.ndim == 2:  # A segmentation with a single frame.
 84        frames = frames[None]
 85
 86    shared_group = seg.SharedFunctionalGroupsSequence[0]
 87    orientation = np.array([float(v) for v in shared_group.PlaneOrientationSequence[0].ImageOrientationPatient])
 88    row_dir, col_dir = orientation[:3], orientation[3:]
 89    normal = np.cross(row_dir, col_dir)
 90    pixel_measures = shared_group.PixelMeasuresSequence[0]
 91    spacing = [float(v) for v in pixel_measures.PixelSpacing]  # The spacing between rows and between columns.
 92
 93    # The frames may be stored in arbitrary order and frames without any foreground may be skipped,
 94    # so the position of each frame along the slice normal is derived from its patient position.
 95    frame_groups = seg.PerFrameFunctionalGroupsSequence
 96    positions = np.array([[float(v) for v in g.PlanePositionSequence[0].ImagePositionPatient] for g in frame_groups])
 97    projections = positions @ normal
 98    if "SpacingBetweenSlices" in pixel_measures:
 99        slice_spacing = float(pixel_measures.SpacingBetweenSlices)
100    elif len(projections) > 1:
101        slice_spacing = np.min(np.diff(np.unique(np.round(projections, 3))))
102    else:
103        slice_spacing = float(pixel_measures.SliceThickness)
104    slice_ids = np.round((projections - projections.min()) / slice_spacing).astype("int")
105
106    # The segment identification is usually per-frame, but for a single-segment file it may instead be
107    # stored once in the shared functional group.
108    shared_segment_number = None
109    if "SegmentIdentificationSequence" in shared_group:
110        shared_segment_number = int(shared_group.SegmentIdentificationSequence[0].ReferencedSegmentNumber)
111
112    labels = np.zeros((slice_ids.max() + 1, seg.Rows, seg.Columns), dtype="uint8")
113    for frame, frame_group, slice_id in zip(frames, frame_groups, slice_ids):
114        if "SegmentIdentificationSequence" in frame_group:
115            segment_number = int(frame_group.SegmentIdentificationSequence[0].ReferencedSegmentNumber)
116        else:
117            segment_number = shared_segment_number
118        labels[slice_id][frame.astype("bool")] = segment_number
119
120    affine = np.eye(4)
121    affine[:3, 0] = normal * slice_spacing
122    affine[:3, 1] = col_dir * spacing[0]
123    affine[:3, 2] = row_dir * spacing[1]
124    affine[:3, 3] = positions[np.argmin(projections)]
125    return labels, affine
126
127
128def _resample_labels(labels, affine, target_shape, target_affine):
129    """Resample a label volume onto the voxel grid of a reference image with nearest neighbor interpolation.
130
131    This is exact if both volumes are stored on the same grid (e.g. a DICOM-SEG object stored on a cropped
132    grid of the reference image) and downsamples segmentations that are stored on a finer grid.
133    """
134    to_label_index = np.linalg.inv(affine) @ target_affine
135    resampled = np.zeros(target_shape, dtype=labels.dtype)
136    yy, xx = np.meshgrid(np.arange(target_shape[1]), np.arange(target_shape[2]), indexing="ij")
137    for z in range(target_shape[0]):
138        target_indices = np.stack([np.full(yy.size, z), yy.ravel(), xx.ravel(), np.ones(yy.size)])
139        indices = np.round(to_label_index[:3] @ target_indices).astype("int")
140        valid = np.all((indices >= 0) & (indices < np.array(labels.shape)[:, None]), axis=0)
141        resampled[z].flat[valid] = labels[tuple(indices[:, valid])]
142    return resampled
143
144
145def _preprocess_adrenal_acc(dicom_dir, csv_path, preprocessed_dir):
146    import h5py
147    import pydicom
148
149    with open(csv_path, "r") as f:
150        seg_series = {row["Subject ID"]: row["Series UID"] for row in csv.DictReader(f) if row["Modality"] == "SEG"}
151
152    os.makedirs(preprocessed_dir, exist_ok=True)
153    for subject_id, seg_uid in tqdm(sorted(seg_series.items()), desc="Preprocess Adrenal-ACC-Ki67-Seg"):
154        out_path = os.path.join(preprocessed_dir, f"{subject_id}.h5")
155        if os.path.exists(out_path):
156            continue
157
158        # The segmentation references the CT series it was created for.
159        seg_path = glob(os.path.join(dicom_dir, seg_uid, "*.dcm"))[0]
160        ct_uid = pydicom.dcmread(seg_path, stop_before_pixels=True).ReferencedSeriesSequence[0].SeriesInstanceUID
161        volume, affine = _load_dicom_volume(os.path.join(dicom_dir, ct_uid))
162        seg_labels, seg_affine = _load_dicom_seg(seg_path)
163        assert seg_labels.max() == 1, f"Expected a single segment in {seg_path}."
164        labels = _resample_labels(seg_labels, seg_affine, volume.shape, affine) * LABEL_IDS["tumor"]
165
166        with h5py.File(out_path, "w") as f:
167            f.create_dataset("raw", data=volume, compression="gzip")
168            f.create_dataset("labels", data=labels, compression="gzip")
169
170
171def get_adrenal_acc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
172    """Download the Adrenal-ACC-Ki67-Seg dataset.
173
174    Args:
175        path: Filepath to a folder where the data is downloaded for further processing.
176        download: Whether to download the data if it is not present.
177
178    Returns:
179        Filepath where the preprocessed data is stored.
180    """
181    preprocessed_dir = os.path.join(path, "preprocessed")
182    if os.path.exists(preprocessed_dir):
183        return preprocessed_dir
184
185    os.makedirs(path, exist_ok=True)
186
187    # Download the DICOM series (CT and SEG) from the TCIA manifest.
188    dicom_dir = os.path.join(path, "dicom")
189    csv_path = os.path.join(path, "adrenal_acc_series")
190    util.download_source_tcia(
191        path=os.path.join(path, "Adrenal-ACC-Ki67-Seg_v1.tcia"), url=URL, dst=dicom_dir,
192        csv_filename=csv_path, download=download,
193    )
194
195    _preprocess_adrenal_acc(dicom_dir, f"{csv_path}.csv", preprocessed_dir)
196    return preprocessed_dir
197
198
199def get_adrenal_acc_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
200    """Get paths to the Adrenal-ACC-Ki67-Seg data.
201
202    Args:
203        path: Filepath to a folder where the data is downloaded for further processing.
204        download: Whether to download the data if it is not present.
205
206    Returns:
207        List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').
208    """
209    data_dir = get_adrenal_acc_data(path, download)
210    volume_paths = natsorted(glob(os.path.join(data_dir, "*.h5")))
211    return volume_paths
212
213
214def get_adrenal_acc_dataset(
215    path: Union[os.PathLike, str],
216    patch_shape: Tuple[int, ...],
217    resize_inputs: bool = False,
218    download: bool = False,
219    **kwargs
220) -> Dataset:
221    """Get the Adrenal-ACC-Ki67-Seg dataset for adrenal tumor segmentation.
222
223    Args:
224        path: Filepath to a folder where the data is downloaded for further processing.
225        patch_shape: The patch shape to use for training.
226        resize_inputs: Whether to resize inputs to the desired patch shape.
227        download: Whether to download the data if it is not present.
228        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
229
230    Returns:
231        The segmentation dataset.
232    """
233    volume_paths = get_adrenal_acc_paths(path, download)
234
235    if resize_inputs:
236        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
237        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
238            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
239        )
240
241    return torch_em.default_segmentation_dataset(
242        raw_paths=volume_paths,
243        raw_key="raw",
244        label_paths=volume_paths,
245        label_key="labels",
246        patch_shape=patch_shape,
247        is_seg_dataset=True,
248        **kwargs
249    )
250
251
252def get_adrenal_acc_loader(
253    path: Union[os.PathLike, str],
254    batch_size: int,
255    patch_shape: Tuple[int, ...],
256    resize_inputs: bool = False,
257    download: bool = False,
258    **kwargs
259) -> DataLoader:
260    """Get the Adrenal-ACC-Ki67-Seg dataloader for adrenal tumor segmentation.
261
262    Args:
263        path: Filepath to a folder where the data is downloaded for further processing.
264        batch_size: The batch size for training.
265        patch_shape: The patch shape to use for training.
266        resize_inputs: Whether to resize inputs to the desired patch shape.
267        download: Whether to download the data if it is not present.
268        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
269
270    Returns:
271        The DataLoader.
272    """
273    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
274    dataset = get_adrenal_acc_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
275    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://www.cancerimagingarchive.net/wp-content/uploads/Adrenal-ACC-Ki67-Seg_v1.tcia'
CHECKSUM = None
LABEL_IDS = {'tumor': 1}
def get_adrenal_acc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
172def get_adrenal_acc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
173    """Download the Adrenal-ACC-Ki67-Seg dataset.
174
175    Args:
176        path: Filepath to a folder where the data is downloaded for further processing.
177        download: Whether to download the data if it is not present.
178
179    Returns:
180        Filepath where the preprocessed data is stored.
181    """
182    preprocessed_dir = os.path.join(path, "preprocessed")
183    if os.path.exists(preprocessed_dir):
184        return preprocessed_dir
185
186    os.makedirs(path, exist_ok=True)
187
188    # Download the DICOM series (CT and SEG) from the TCIA manifest.
189    dicom_dir = os.path.join(path, "dicom")
190    csv_path = os.path.join(path, "adrenal_acc_series")
191    util.download_source_tcia(
192        path=os.path.join(path, "Adrenal-ACC-Ki67-Seg_v1.tcia"), url=URL, dst=dicom_dir,
193        csv_filename=csv_path, download=download,
194    )
195
196    _preprocess_adrenal_acc(dicom_dir, f"{csv_path}.csv", preprocessed_dir)
197    return preprocessed_dir

Download the Adrenal-ACC-Ki67-Seg dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the preprocessed data is stored.

def get_adrenal_acc_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
200def get_adrenal_acc_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]:
201    """Get paths to the Adrenal-ACC-Ki67-Seg data.
202
203    Args:
204        path: Filepath to a folder where the data is downloaded for further processing.
205        download: Whether to download the data if it is not present.
206
207    Returns:
208        List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').
209    """
210    data_dir = get_adrenal_acc_data(path, download)
211    volume_paths = natsorted(glob(os.path.join(data_dir, "*.h5")))
212    return volume_paths

Get paths to the Adrenal-ACC-Ki67-Seg data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').

def get_adrenal_acc_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
215def get_adrenal_acc_dataset(
216    path: Union[os.PathLike, str],
217    patch_shape: Tuple[int, ...],
218    resize_inputs: bool = False,
219    download: bool = False,
220    **kwargs
221) -> Dataset:
222    """Get the Adrenal-ACC-Ki67-Seg dataset for adrenal tumor segmentation.
223
224    Args:
225        path: Filepath to a folder where the data is downloaded for further processing.
226        patch_shape: The patch shape to use for training.
227        resize_inputs: Whether to resize inputs to the desired patch shape.
228        download: Whether to download the data if it is not present.
229        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
230
231    Returns:
232        The segmentation dataset.
233    """
234    volume_paths = get_adrenal_acc_paths(path, download)
235
236    if resize_inputs:
237        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
238        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
239            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
240        )
241
242    return torch_em.default_segmentation_dataset(
243        raw_paths=volume_paths,
244        raw_key="raw",
245        label_paths=volume_paths,
246        label_key="labels",
247        patch_shape=patch_shape,
248        is_seg_dataset=True,
249        **kwargs
250    )

Get the Adrenal-ACC-Ki67-Seg dataset for adrenal tumor segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_adrenal_acc_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
253def get_adrenal_acc_loader(
254    path: Union[os.PathLike, str],
255    batch_size: int,
256    patch_shape: Tuple[int, ...],
257    resize_inputs: bool = False,
258    download: bool = False,
259    **kwargs
260) -> DataLoader:
261    """Get the Adrenal-ACC-Ki67-Seg dataloader for adrenal tumor segmentation.
262
263    Args:
264        path: Filepath to a folder where the data is downloaded for further processing.
265        batch_size: The batch size for training.
266        patch_shape: The patch shape to use for training.
267        resize_inputs: Whether to resize inputs to the desired patch shape.
268        download: Whether to download the data if it is not present.
269        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
270
271    Returns:
272        The DataLoader.
273    """
274    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
275    dataset = get_adrenal_acc_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
276    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the Adrenal-ACC-Ki67-Seg dataloader for adrenal tumor segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.