torch_em.data.datasets.medical.adrenal_acc
The Adrenal-ACC-Ki67-Seg dataset contains annotations for adrenocortical carcinoma segmentation in contrast-enhanced abdominal CT.
It consists of 53 CT volumes (one segmented CT series per patient, the collection also contains 71 additional unsegmented CT series of the same patients) with binary tumor labels (1: adrenal tumor). The CT scans are distributed as DICOM series and the labels as DICOM-SEG objects, which are converted and stored in hdf5 files by this module.
NOTE: This requires the pydicom python package.
The dataset is located at https://www.cancerimagingarchive.net/collection/adrenal-acc-ki67-seg/.
This dataset is from the publication https://doi.org/10.1016/j.crad.2020.01.012. The data was released at https://doi.org/10.7937/1FPG-VM46. Please cite it if you use this dataset in your research.
1"""The Adrenal-ACC-Ki67-Seg dataset contains annotations for adrenocortical carcinoma segmentation 2in contrast-enhanced abdominal CT. 3 4It consists of 53 CT volumes (one segmented CT series per patient, the collection also contains 71 additional 5unsegmented CT series of the same patients) with binary tumor labels (1: adrenal tumor). The CT scans are 6distributed as DICOM series and the labels as DICOM-SEG objects, which are converted and stored in hdf5 files 7by this module. 8 9NOTE: This requires the pydicom python package. 10 11The dataset is located at https://www.cancerimagingarchive.net/collection/adrenal-acc-ki67-seg/. 12 13This dataset is from the publication https://doi.org/10.1016/j.crad.2020.01.012. 14The data was released at https://doi.org/10.7937/1FPG-VM46. 15Please cite it if you use this dataset in your research. 16""" 17 18import os 19import csv 20from glob import glob 21from tqdm import tqdm 22from natsort import natsorted 23from typing import Union, Tuple, List 24 25import numpy as np 26 27from torch.utils.data import Dataset, DataLoader 28 29import torch_em 30 31from .. import util 32 33 34URL = "https://www.cancerimagingarchive.net/wp-content/uploads/Adrenal-ACC-Ki67-Seg_v1.tcia" 35 36# The DICOM series are downloaded individually from TCIA. 37CHECKSUM = None 38 39LABEL_IDS = {"tumor": 1} 40 41 42def _load_dicom_volume(series_dir): 43 """Stack a DICOM series into a volume with axes (z, y, x) and slices sorted along the slice normal. 44 45 Applies the rescale slope/intercept if present (e.g. to get Hounsfield units for CT); MR series 46 usually do not carry these tags, in which case the stored pixel values are used as-is, per the 47 DICOM standard's own default of a slope of 1 and an intercept of 0. 48 49 Returns the volume and the affine matrix that maps voxel indices (z, y, x) to DICOM patient 50 coordinates. 51 """ 52 import pydicom 53 54 slices = [pydicom.dcmread(dcm_path) for dcm_path in natsorted(glob(os.path.join(series_dir, "*.dcm")))] 55 orientation = np.array([float(v) for v in slices[0].ImageOrientationPatient]) 56 row_dir, col_dir = orientation[:3], orientation[3:] 57 normal = np.cross(row_dir, col_dir) 58 slices.sort(key=lambda dcm: np.dot([float(v) for v in dcm.ImagePositionPatient], normal)) 59 60 volume = np.stack([dcm.pixel_array for dcm in slices]).astype("float32") 61 volume = volume * float(getattr(slices[0], "RescaleSlope", 1.0)) + float(getattr(slices[0], "RescaleIntercept", 0.0)) # noqa 62 volume = np.round(volume).astype("int16") 63 64 positions = np.array([[float(v) for v in dcm.ImagePositionPatient] for dcm in slices]) 65 spacing = [float(v) for v in slices[0].PixelSpacing] # The spacing between rows and between columns. 66 affine = np.eye(4) 67 affine[:3, 0] = (positions[-1] - positions[0]) / (len(slices) - 1) 68 affine[:3, 1] = col_dir * spacing[0] 69 affine[:3, 2] = row_dir * spacing[1] 70 affine[:3, 3] = positions[0] 71 return volume, affine 72 73 74def _load_dicom_seg(seg_path): 75 """Load a DICOM-SEG object as a label volume with axes (z, y, x), where the segment number is used as label id. 76 77 Returns the label volume and the affine matrix that maps its voxel indices to DICOM patient coordinates. 78 """ 79 import pydicom 80 81 seg = pydicom.dcmread(seg_path) 82 frames = seg.pixel_array 83 if frames.ndim == 2: # A segmentation with a single frame. 84 frames = frames[None] 85 86 shared_group = seg.SharedFunctionalGroupsSequence[0] 87 orientation = np.array([float(v) for v in shared_group.PlaneOrientationSequence[0].ImageOrientationPatient]) 88 row_dir, col_dir = orientation[:3], orientation[3:] 89 normal = np.cross(row_dir, col_dir) 90 pixel_measures = shared_group.PixelMeasuresSequence[0] 91 spacing = [float(v) for v in pixel_measures.PixelSpacing] # The spacing between rows and between columns. 92 93 # The frames may be stored in arbitrary order and frames without any foreground may be skipped, 94 # so the position of each frame along the slice normal is derived from its patient position. 95 frame_groups = seg.PerFrameFunctionalGroupsSequence 96 positions = np.array([[float(v) for v in g.PlanePositionSequence[0].ImagePositionPatient] for g in frame_groups]) 97 projections = positions @ normal 98 if "SpacingBetweenSlices" in pixel_measures: 99 slice_spacing = float(pixel_measures.SpacingBetweenSlices) 100 elif len(projections) > 1: 101 slice_spacing = np.min(np.diff(np.unique(np.round(projections, 3)))) 102 else: 103 slice_spacing = float(pixel_measures.SliceThickness) 104 slice_ids = np.round((projections - projections.min()) / slice_spacing).astype("int") 105 106 # The segment identification is usually per-frame, but for a single-segment file it may instead be 107 # stored once in the shared functional group. 108 shared_segment_number = None 109 if "SegmentIdentificationSequence" in shared_group: 110 shared_segment_number = int(shared_group.SegmentIdentificationSequence[0].ReferencedSegmentNumber) 111 112 labels = np.zeros((slice_ids.max() + 1, seg.Rows, seg.Columns), dtype="uint8") 113 for frame, frame_group, slice_id in zip(frames, frame_groups, slice_ids): 114 if "SegmentIdentificationSequence" in frame_group: 115 segment_number = int(frame_group.SegmentIdentificationSequence[0].ReferencedSegmentNumber) 116 else: 117 segment_number = shared_segment_number 118 labels[slice_id][frame.astype("bool")] = segment_number 119 120 affine = np.eye(4) 121 affine[:3, 0] = normal * slice_spacing 122 affine[:3, 1] = col_dir * spacing[0] 123 affine[:3, 2] = row_dir * spacing[1] 124 affine[:3, 3] = positions[np.argmin(projections)] 125 return labels, affine 126 127 128def _resample_labels(labels, affine, target_shape, target_affine): 129 """Resample a label volume onto the voxel grid of a reference image with nearest neighbor interpolation. 130 131 This is exact if both volumes are stored on the same grid (e.g. a DICOM-SEG object stored on a cropped 132 grid of the reference image) and downsamples segmentations that are stored on a finer grid. 133 """ 134 to_label_index = np.linalg.inv(affine) @ target_affine 135 resampled = np.zeros(target_shape, dtype=labels.dtype) 136 yy, xx = np.meshgrid(np.arange(target_shape[1]), np.arange(target_shape[2]), indexing="ij") 137 for z in range(target_shape[0]): 138 target_indices = np.stack([np.full(yy.size, z), yy.ravel(), xx.ravel(), np.ones(yy.size)]) 139 indices = np.round(to_label_index[:3] @ target_indices).astype("int") 140 valid = np.all((indices >= 0) & (indices < np.array(labels.shape)[:, None]), axis=0) 141 resampled[z].flat[valid] = labels[tuple(indices[:, valid])] 142 return resampled 143 144 145def _preprocess_adrenal_acc(dicom_dir, csv_path, preprocessed_dir): 146 import h5py 147 import pydicom 148 149 with open(csv_path, "r") as f: 150 seg_series = {row["Subject ID"]: row["Series UID"] for row in csv.DictReader(f) if row["Modality"] == "SEG"} 151 152 os.makedirs(preprocessed_dir, exist_ok=True) 153 for subject_id, seg_uid in tqdm(sorted(seg_series.items()), desc="Preprocess Adrenal-ACC-Ki67-Seg"): 154 out_path = os.path.join(preprocessed_dir, f"{subject_id}.h5") 155 if os.path.exists(out_path): 156 continue 157 158 # The segmentation references the CT series it was created for. 159 seg_path = glob(os.path.join(dicom_dir, seg_uid, "*.dcm"))[0] 160 ct_uid = pydicom.dcmread(seg_path, stop_before_pixels=True).ReferencedSeriesSequence[0].SeriesInstanceUID 161 volume, affine = _load_dicom_volume(os.path.join(dicom_dir, ct_uid)) 162 seg_labels, seg_affine = _load_dicom_seg(seg_path) 163 assert seg_labels.max() == 1, f"Expected a single segment in {seg_path}." 164 labels = _resample_labels(seg_labels, seg_affine, volume.shape, affine) * LABEL_IDS["tumor"] 165 166 with h5py.File(out_path, "w") as f: 167 f.create_dataset("raw", data=volume, compression="gzip") 168 f.create_dataset("labels", data=labels, compression="gzip") 169 170 171def get_adrenal_acc_data(path: Union[os.PathLike, str], download: bool = False) -> str: 172 """Download the Adrenal-ACC-Ki67-Seg dataset. 173 174 Args: 175 path: Filepath to a folder where the data is downloaded for further processing. 176 download: Whether to download the data if it is not present. 177 178 Returns: 179 Filepath where the preprocessed data is stored. 180 """ 181 preprocessed_dir = os.path.join(path, "preprocessed") 182 if os.path.exists(preprocessed_dir): 183 return preprocessed_dir 184 185 os.makedirs(path, exist_ok=True) 186 187 # Download the DICOM series (CT and SEG) from the TCIA manifest. 188 dicom_dir = os.path.join(path, "dicom") 189 csv_path = os.path.join(path, "adrenal_acc_series") 190 util.download_source_tcia( 191 path=os.path.join(path, "Adrenal-ACC-Ki67-Seg_v1.tcia"), url=URL, dst=dicom_dir, 192 csv_filename=csv_path, download=download, 193 ) 194 195 _preprocess_adrenal_acc(dicom_dir, f"{csv_path}.csv", preprocessed_dir) 196 return preprocessed_dir 197 198 199def get_adrenal_acc_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]: 200 """Get paths to the Adrenal-ACC-Ki67-Seg data. 201 202 Args: 203 path: Filepath to a folder where the data is downloaded for further processing. 204 download: Whether to download the data if it is not present. 205 206 Returns: 207 List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels'). 208 """ 209 data_dir = get_adrenal_acc_data(path, download) 210 volume_paths = natsorted(glob(os.path.join(data_dir, "*.h5"))) 211 return volume_paths 212 213 214def get_adrenal_acc_dataset( 215 path: Union[os.PathLike, str], 216 patch_shape: Tuple[int, ...], 217 resize_inputs: bool = False, 218 download: bool = False, 219 **kwargs 220) -> Dataset: 221 """Get the Adrenal-ACC-Ki67-Seg dataset for adrenal tumor segmentation. 222 223 Args: 224 path: Filepath to a folder where the data is downloaded for further processing. 225 patch_shape: The patch shape to use for training. 226 resize_inputs: Whether to resize inputs to the desired patch shape. 227 download: Whether to download the data if it is not present. 228 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 229 230 Returns: 231 The segmentation dataset. 232 """ 233 volume_paths = get_adrenal_acc_paths(path, download) 234 235 if resize_inputs: 236 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 237 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 238 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 239 ) 240 241 return torch_em.default_segmentation_dataset( 242 raw_paths=volume_paths, 243 raw_key="raw", 244 label_paths=volume_paths, 245 label_key="labels", 246 patch_shape=patch_shape, 247 is_seg_dataset=True, 248 **kwargs 249 ) 250 251 252def get_adrenal_acc_loader( 253 path: Union[os.PathLike, str], 254 batch_size: int, 255 patch_shape: Tuple[int, ...], 256 resize_inputs: bool = False, 257 download: bool = False, 258 **kwargs 259) -> DataLoader: 260 """Get the Adrenal-ACC-Ki67-Seg dataloader for adrenal tumor segmentation. 261 262 Args: 263 path: Filepath to a folder where the data is downloaded for further processing. 264 batch_size: The batch size for training. 265 patch_shape: The patch shape to use for training. 266 resize_inputs: Whether to resize inputs to the desired patch shape. 267 download: Whether to download the data if it is not present. 268 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 269 270 Returns: 271 The DataLoader. 272 """ 273 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 274 dataset = get_adrenal_acc_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 275 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
172def get_adrenal_acc_data(path: Union[os.PathLike, str], download: bool = False) -> str: 173 """Download the Adrenal-ACC-Ki67-Seg dataset. 174 175 Args: 176 path: Filepath to a folder where the data is downloaded for further processing. 177 download: Whether to download the data if it is not present. 178 179 Returns: 180 Filepath where the preprocessed data is stored. 181 """ 182 preprocessed_dir = os.path.join(path, "preprocessed") 183 if os.path.exists(preprocessed_dir): 184 return preprocessed_dir 185 186 os.makedirs(path, exist_ok=True) 187 188 # Download the DICOM series (CT and SEG) from the TCIA manifest. 189 dicom_dir = os.path.join(path, "dicom") 190 csv_path = os.path.join(path, "adrenal_acc_series") 191 util.download_source_tcia( 192 path=os.path.join(path, "Adrenal-ACC-Ki67-Seg_v1.tcia"), url=URL, dst=dicom_dir, 193 csv_filename=csv_path, download=download, 194 ) 195 196 _preprocess_adrenal_acc(dicom_dir, f"{csv_path}.csv", preprocessed_dir) 197 return preprocessed_dir
Download the Adrenal-ACC-Ki67-Seg dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the preprocessed data is stored.
200def get_adrenal_acc_paths(path: Union[os.PathLike, str], download: bool = False) -> List[str]: 201 """Get paths to the Adrenal-ACC-Ki67-Seg data. 202 203 Args: 204 path: Filepath to a folder where the data is downloaded for further processing. 205 download: Whether to download the data if it is not present. 206 207 Returns: 208 List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels'). 209 """ 210 data_dir = get_adrenal_acc_data(path, download) 211 volume_paths = natsorted(glob(os.path.join(data_dir, "*.h5"))) 212 return volume_paths
Get paths to the Adrenal-ACC-Ki67-Seg data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels').
215def get_adrenal_acc_dataset( 216 path: Union[os.PathLike, str], 217 patch_shape: Tuple[int, ...], 218 resize_inputs: bool = False, 219 download: bool = False, 220 **kwargs 221) -> Dataset: 222 """Get the Adrenal-ACC-Ki67-Seg dataset for adrenal tumor segmentation. 223 224 Args: 225 path: Filepath to a folder where the data is downloaded for further processing. 226 patch_shape: The patch shape to use for training. 227 resize_inputs: Whether to resize inputs to the desired patch shape. 228 download: Whether to download the data if it is not present. 229 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 230 231 Returns: 232 The segmentation dataset. 233 """ 234 volume_paths = get_adrenal_acc_paths(path, download) 235 236 if resize_inputs: 237 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 238 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 239 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 240 ) 241 242 return torch_em.default_segmentation_dataset( 243 raw_paths=volume_paths, 244 raw_key="raw", 245 label_paths=volume_paths, 246 label_key="labels", 247 patch_shape=patch_shape, 248 is_seg_dataset=True, 249 **kwargs 250 )
Get the Adrenal-ACC-Ki67-Seg dataset for adrenal tumor segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
253def get_adrenal_acc_loader( 254 path: Union[os.PathLike, str], 255 batch_size: int, 256 patch_shape: Tuple[int, ...], 257 resize_inputs: bool = False, 258 download: bool = False, 259 **kwargs 260) -> DataLoader: 261 """Get the Adrenal-ACC-Ki67-Seg dataloader for adrenal tumor segmentation. 262 263 Args: 264 path: Filepath to a folder where the data is downloaded for further processing. 265 batch_size: The batch size for training. 266 patch_shape: The patch shape to use for training. 267 resize_inputs: Whether to resize inputs to the desired patch shape. 268 download: Whether to download the data if it is not present. 269 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 270 271 Returns: 272 The DataLoader. 273 """ 274 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 275 dataset = get_adrenal_acc_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 276 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the Adrenal-ACC-Ki67-Seg dataloader for adrenal tumor segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.