torch_em.data.datasets.medical.hqcolon
HQColon is a clinically validated dataset of 435 human colons segmented from CT colonography (CTC).
The CTC volumes are from the publicly available CT Colonography collection on The Cancer Imaging Archive (TCIA), and are downloaded here directly from TCIA by their series instance UID. For each volume, two segmentation masks are provided: one for the entire colon (including collapsed segments and fluid) and one for only the gas-filled parts of the colon. Both masks were generated with a hybrid interactive machine learning pipeline and clinically validated by an expert abdominal radiologist.
NOTE: This requires the pydicom python package.
The dataset is located at https://doi.org/10.17605/OSF.IO/8TKPM.
This dataset is from the publications https://doi.org/10.1038/s41597-025-06518-z (dataset) and https://doi.org/10.48550/arXiv.2502.21183 (annotation pipeline). Please cite them if you use this dataset in your research.
1"""HQColon is a clinically validated dataset of 435 human colons segmented from CT colonography (CTC). 2 3The CTC volumes are from the publicly available CT Colonography collection on The Cancer Imaging 4Archive (TCIA), and are downloaded here directly from TCIA by their series instance UID. For each 5volume, two segmentation masks are provided: one for the entire colon (including collapsed segments 6and fluid) and one for only the gas-filled parts of the colon. Both masks were generated with a 7hybrid interactive machine learning pipeline and clinically validated by an expert abdominal 8radiologist. 9 10NOTE: This requires the pydicom python package. 11 12The dataset is located at https://doi.org/10.17605/OSF.IO/8TKPM. 13 14This dataset is from the publications https://doi.org/10.1038/s41597-025-06518-z (dataset) and 15https://doi.org/10.48550/arXiv.2502.21183 (annotation pipeline). Please cite them if you use this 16dataset in your research. 17""" 18 19import os 20import json 21from glob import glob 22from tqdm import tqdm 23from natsort import natsorted 24from typing import Union, Tuple, List, Literal 25 26from torch.utils.data import Dataset, DataLoader 27 28import torch_em 29 30from .. import util 31 32 33URLS = { 34 "metadata": "https://osf.io/download/8w6q7/", 35 "gas_and_fluid": "https://osf.io/download/d4sc3/", 36 "gas": "https://osf.io/download/y3ad2/", 37} 38 39CHECKSUMS = { 40 "metadata": "158bd6b4551c07f60ba3d32c7702ef67165b03308a5e1b5fa9e943598dd77693", 41 "gas_and_fluid": "99c0986b03291dbd0d4d973dc35bc5900e575fdb2f9ac9ea584381a5b12240bc", 42 "gas": "04bcb14aec9c4734756853f7c4b439b7de3c4a2ee30cedf482939a633cd4d840", 43} 44 45MASK_FOLDERS = {"gas_and_fluid": "Segmentation Air and Fluid", "gas": "Segmentation Air"} 46 47 48def _load_entries(metadata_path): 49 with open(metadata_path, "r") as f: 50 return [json.loads(line) for line in f if line.strip()] 51 52 53def _preprocess_hqcolon(path, entries, dicom_dir, preprocessed_dir): 54 import SimpleITK as sitk 55 56 os.makedirs(preprocessed_dir, exist_ok=True) 57 for entry in tqdm(entries, desc="Preprocess HQColon"): 58 out_path = os.path.join(preprocessed_dir, f"{entry['subject_id']}.h5") 59 if os.path.exists(out_path): 60 continue 61 62 series_dir = os.path.join(dicom_dir, entry["InstanceUID"]) 63 if not glob(os.path.join(series_dir, "*.dcm")): 64 continue 65 66 gas_fluid_path = os.path.join(path, MASK_FOLDERS["gas_and_fluid"], entry["nnunet_label_file"]) 67 gas_path = os.path.join(path, MASK_FOLDERS["gas"], entry["nnunet_label_file"]) 68 if not (os.path.exists(gas_fluid_path) and os.path.exists(gas_path)): 69 continue 70 71 volume, _ = util.load_dicom_series(series_dir) 72 labels_gas_fluid = sitk.GetArrayFromImage(sitk.ReadImage(gas_fluid_path)) 73 labels_gas = sitk.GetArrayFromImage(sitk.ReadImage(gas_path)) 74 75 assert volume.shape == labels_gas_fluid.shape == labels_gas.shape, \ 76 f"Shape mismatch for {entry['subject_id']}: {volume.shape}, {labels_gas_fluid.shape}, {labels_gas.shape}" 77 78 import h5py 79 with h5py.File(out_path, "w") as f: 80 f.create_dataset("raw", data=volume, compression="gzip") 81 f.create_dataset("labels/gas_and_fluid", data=labels_gas_fluid.astype("uint8"), compression="gzip") 82 f.create_dataset("labels/gas", data=labels_gas.astype("uint8"), compression="gzip") 83 84 85def get_hqcolon_data(path: Union[os.PathLike, str], download: bool = False) -> str: 86 """Download the HQColon dataset. 87 88 Args: 89 path: Filepath to a folder where the data is downloaded for further processing. 90 download: Whether to download the data if it is not present. 91 92 Returns: 93 Filepath where the preprocessed data is stored. 94 """ 95 # NOTE: The preprocessing below skips volumes that were converted already, so an interrupted run resumes. 96 preprocessed_dir = os.path.join(path, "preprocessed") 97 98 os.makedirs(path, exist_ok=True) 99 100 metadata_path = os.path.join(path, "meta-data.json") 101 util.download_source(path=metadata_path, url=URLS["metadata"], download=download, checksum=CHECKSUMS["metadata"]) 102 entries = _load_entries(metadata_path) 103 104 for name in ["gas_and_fluid", "gas"]: 105 mask_dir = os.path.join(path, MASK_FOLDERS[name]) 106 if os.path.exists(mask_dir): 107 continue 108 zip_path = os.path.join(path, f"{name}.zip") 109 util.download_source(path=zip_path, url=URLS[name], download=download, checksum=CHECKSUMS[name]) 110 util.unzip(zip_path=zip_path, dst=path) 111 112 dicom_dir = os.path.join(path, "dicom") 113 if download: 114 series_uids = [entry["InstanceUID"] for entry in entries] 115 util.download_tcia_series(series_uids, dst=dicom_dir, csv_filename=os.path.join(path, "hqcolon_series")) 116 117 _preprocess_hqcolon(path, entries, dicom_dir, preprocessed_dir) 118 return preprocessed_dir 119 120 121def get_hqcolon_paths( 122 path: Union[os.PathLike, str], 123 label_choice: Literal["gas_and_fluid", "gas"] = "gas_and_fluid", 124 download: bool = False, 125) -> List[str]: 126 """Get paths to the HQColon data. 127 128 Args: 129 path: Filepath to a folder where the data is downloaded for further processing. 130 label_choice: The choice of segmentation mask. Either 'gas_and_fluid' (the entire colon, including 131 collapsed segments and fluid) or 'gas' (only the gas-filled parts of the colon). 132 download: Whether to download the data if it is not present. 133 134 Returns: 135 List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data 136 ('labels/gas_and_fluid' and 'labels/gas'). 137 """ 138 if label_choice not in MASK_FOLDERS: 139 raise ValueError(f"'{label_choice}' is not a valid label choice. Choose from {list(MASK_FOLDERS.keys())}.") 140 141 preprocessed_dir = get_hqcolon_data(path, download) 142 volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5"))) 143 assert len(volume_paths) > 0, f"Could not find any preprocessed samples in '{preprocessed_dir}'." 144 return volume_paths 145 146 147def get_hqcolon_dataset( 148 path: Union[os.PathLike, str], 149 patch_shape: Tuple[int, ...], 150 label_choice: Literal["gas_and_fluid", "gas"] = "gas_and_fluid", 151 resize_inputs: bool = False, 152 download: bool = False, 153 **kwargs 154) -> Dataset: 155 """Get the HQColon dataset for colon segmentation in CT colonography. 156 157 Args: 158 path: Filepath to a folder where the data is downloaded for further processing. 159 patch_shape: The patch shape to use for training. 160 label_choice: The choice of segmentation mask. Either 'gas_and_fluid' or 'gas'. 161 resize_inputs: Whether to resize inputs to the desired patch shape. 162 download: Whether to download the data if it is not present. 163 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 164 165 Returns: 166 The segmentation dataset. 167 """ 168 volume_paths = get_hqcolon_paths(path, label_choice, download) 169 170 if resize_inputs: 171 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 172 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 173 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 174 ) 175 176 return torch_em.default_segmentation_dataset( 177 raw_paths=volume_paths, 178 raw_key="raw", 179 label_paths=volume_paths, 180 label_key=f"labels/{label_choice}", 181 patch_shape=patch_shape, 182 is_seg_dataset=True, 183 **kwargs 184 ) 185 186 187def get_hqcolon_loader( 188 path: Union[os.PathLike, str], 189 batch_size: int, 190 patch_shape: Tuple[int, ...], 191 label_choice: Literal["gas_and_fluid", "gas"] = "gas_and_fluid", 192 resize_inputs: bool = False, 193 download: bool = False, 194 **kwargs 195) -> DataLoader: 196 """Get the HQColon dataloader for colon segmentation in CT colonography. 197 198 Args: 199 path: Filepath to a folder where the data is downloaded for further processing. 200 batch_size: The batch size for training. 201 patch_shape: The patch shape to use for training. 202 label_choice: The choice of segmentation mask. Either 'gas_and_fluid' or 'gas'. 203 resize_inputs: Whether to resize inputs to the desired patch shape. 204 download: Whether to download the data if it is not present. 205 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 206 207 Returns: 208 The DataLoader. 209 """ 210 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 211 dataset = get_hqcolon_dataset(path, patch_shape, label_choice, resize_inputs, download, **ds_kwargs) 212 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
86def get_hqcolon_data(path: Union[os.PathLike, str], download: bool = False) -> str: 87 """Download the HQColon dataset. 88 89 Args: 90 path: Filepath to a folder where the data is downloaded for further processing. 91 download: Whether to download the data if it is not present. 92 93 Returns: 94 Filepath where the preprocessed data is stored. 95 """ 96 # NOTE: The preprocessing below skips volumes that were converted already, so an interrupted run resumes. 97 preprocessed_dir = os.path.join(path, "preprocessed") 98 99 os.makedirs(path, exist_ok=True) 100 101 metadata_path = os.path.join(path, "meta-data.json") 102 util.download_source(path=metadata_path, url=URLS["metadata"], download=download, checksum=CHECKSUMS["metadata"]) 103 entries = _load_entries(metadata_path) 104 105 for name in ["gas_and_fluid", "gas"]: 106 mask_dir = os.path.join(path, MASK_FOLDERS[name]) 107 if os.path.exists(mask_dir): 108 continue 109 zip_path = os.path.join(path, f"{name}.zip") 110 util.download_source(path=zip_path, url=URLS[name], download=download, checksum=CHECKSUMS[name]) 111 util.unzip(zip_path=zip_path, dst=path) 112 113 dicom_dir = os.path.join(path, "dicom") 114 if download: 115 series_uids = [entry["InstanceUID"] for entry in entries] 116 util.download_tcia_series(series_uids, dst=dicom_dir, csv_filename=os.path.join(path, "hqcolon_series")) 117 118 _preprocess_hqcolon(path, entries, dicom_dir, preprocessed_dir) 119 return preprocessed_dir
Download the HQColon dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the preprocessed data is stored.
122def get_hqcolon_paths( 123 path: Union[os.PathLike, str], 124 label_choice: Literal["gas_and_fluid", "gas"] = "gas_and_fluid", 125 download: bool = False, 126) -> List[str]: 127 """Get paths to the HQColon data. 128 129 Args: 130 path: Filepath to a folder where the data is downloaded for further processing. 131 label_choice: The choice of segmentation mask. Either 'gas_and_fluid' (the entire colon, including 132 collapsed segments and fluid) or 'gas' (only the gas-filled parts of the colon). 133 download: Whether to download the data if it is not present. 134 135 Returns: 136 List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data 137 ('labels/gas_and_fluid' and 'labels/gas'). 138 """ 139 if label_choice not in MASK_FOLDERS: 140 raise ValueError(f"'{label_choice}' is not a valid label choice. Choose from {list(MASK_FOLDERS.keys())}.") 141 142 preprocessed_dir = get_hqcolon_data(path, download) 143 volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5"))) 144 assert len(volume_paths) > 0, f"Could not find any preprocessed samples in '{preprocessed_dir}'." 145 return volume_paths
Get paths to the HQColon data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- label_choice: The choice of segmentation mask. Either 'gas_and_fluid' (the entire colon, including collapsed segments and fluid) or 'gas' (only the gas-filled parts of the colon).
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels/gas_and_fluid' and 'labels/gas').
148def get_hqcolon_dataset( 149 path: Union[os.PathLike, str], 150 patch_shape: Tuple[int, ...], 151 label_choice: Literal["gas_and_fluid", "gas"] = "gas_and_fluid", 152 resize_inputs: bool = False, 153 download: bool = False, 154 **kwargs 155) -> Dataset: 156 """Get the HQColon dataset for colon segmentation in CT colonography. 157 158 Args: 159 path: Filepath to a folder where the data is downloaded for further processing. 160 patch_shape: The patch shape to use for training. 161 label_choice: The choice of segmentation mask. Either 'gas_and_fluid' or 'gas'. 162 resize_inputs: Whether to resize inputs to the desired patch shape. 163 download: Whether to download the data if it is not present. 164 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 165 166 Returns: 167 The segmentation dataset. 168 """ 169 volume_paths = get_hqcolon_paths(path, label_choice, download) 170 171 if resize_inputs: 172 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 173 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 174 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 175 ) 176 177 return torch_em.default_segmentation_dataset( 178 raw_paths=volume_paths, 179 raw_key="raw", 180 label_paths=volume_paths, 181 label_key=f"labels/{label_choice}", 182 patch_shape=patch_shape, 183 is_seg_dataset=True, 184 **kwargs 185 )
Get the HQColon dataset for colon segmentation in CT colonography.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- label_choice: The choice of segmentation mask. Either 'gas_and_fluid' or 'gas'.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
188def get_hqcolon_loader( 189 path: Union[os.PathLike, str], 190 batch_size: int, 191 patch_shape: Tuple[int, ...], 192 label_choice: Literal["gas_and_fluid", "gas"] = "gas_and_fluid", 193 resize_inputs: bool = False, 194 download: bool = False, 195 **kwargs 196) -> DataLoader: 197 """Get the HQColon dataloader for colon segmentation in CT colonography. 198 199 Args: 200 path: Filepath to a folder where the data is downloaded for further processing. 201 batch_size: The batch size for training. 202 patch_shape: The patch shape to use for training. 203 label_choice: The choice of segmentation mask. Either 'gas_and_fluid' or 'gas'. 204 resize_inputs: Whether to resize inputs to the desired patch shape. 205 download: Whether to download the data if it is not present. 206 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 207 208 Returns: 209 The DataLoader. 210 """ 211 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 212 dataset = get_hqcolon_dataset(path, patch_shape, label_choice, resize_inputs, download, **ds_kwargs) 213 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the HQColon dataloader for colon segmentation in CT colonography.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- label_choice: The choice of segmentation mask. Either 'gas_and_fluid' or 'gas'.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.