torch_em.data.datasets.medical.hqcolon

HQColon is a clinically validated dataset of 435 human colons segmented from CT colonography (CTC).

The CTC volumes are from the publicly available CT Colonography collection on The Cancer Imaging Archive (TCIA), and are downloaded here directly from TCIA by their series instance UID. For each volume, two segmentation masks are provided: one for the entire colon (including collapsed segments and fluid) and one for only the gas-filled parts of the colon. Both masks were generated with a hybrid interactive machine learning pipeline and clinically validated by an expert abdominal radiologist.

NOTE: This requires the pydicom python package.

The dataset is located at https://doi.org/10.17605/OSF.IO/8TKPM.

This dataset is from the publications https://doi.org/10.1038/s41597-025-06518-z (dataset) and https://doi.org/10.48550/arXiv.2502.21183 (annotation pipeline). Please cite them if you use this dataset in your research.

  1"""HQColon is a clinically validated dataset of 435 human colons segmented from CT colonography (CTC).
  2
  3The CTC volumes are from the publicly available CT Colonography collection on The Cancer Imaging
  4Archive (TCIA), and are downloaded here directly from TCIA by their series instance UID. For each
  5volume, two segmentation masks are provided: one for the entire colon (including collapsed segments
  6and fluid) and one for only the gas-filled parts of the colon. Both masks were generated with a
  7hybrid interactive machine learning pipeline and clinically validated by an expert abdominal
  8radiologist.
  9
 10NOTE: This requires the pydicom python package.
 11
 12The dataset is located at https://doi.org/10.17605/OSF.IO/8TKPM.
 13
 14This dataset is from the publications https://doi.org/10.1038/s41597-025-06518-z (dataset) and
 15https://doi.org/10.48550/arXiv.2502.21183 (annotation pipeline). Please cite them if you use this
 16dataset in your research.
 17"""
 18
 19import os
 20import json
 21from glob import glob
 22from tqdm import tqdm
 23from natsort import natsorted
 24from typing import Union, Tuple, List, Literal
 25
 26from torch.utils.data import Dataset, DataLoader
 27
 28import torch_em
 29
 30from .. import util
 31
 32
 33URLS = {
 34    "metadata": "https://osf.io/download/8w6q7/",
 35    "gas_and_fluid": "https://osf.io/download/d4sc3/",
 36    "gas": "https://osf.io/download/y3ad2/",
 37}
 38
 39CHECKSUMS = {
 40    "metadata": "158bd6b4551c07f60ba3d32c7702ef67165b03308a5e1b5fa9e943598dd77693",
 41    "gas_and_fluid": "99c0986b03291dbd0d4d973dc35bc5900e575fdb2f9ac9ea584381a5b12240bc",
 42    "gas": "04bcb14aec9c4734756853f7c4b439b7de3c4a2ee30cedf482939a633cd4d840",
 43}
 44
 45MASK_FOLDERS = {"gas_and_fluid": "Segmentation Air and Fluid", "gas": "Segmentation Air"}
 46
 47
 48def _load_entries(metadata_path):
 49    with open(metadata_path, "r") as f:
 50        return [json.loads(line) for line in f if line.strip()]
 51
 52
 53def _preprocess_hqcolon(path, entries, dicom_dir, preprocessed_dir):
 54    import SimpleITK as sitk
 55
 56    os.makedirs(preprocessed_dir, exist_ok=True)
 57    for entry in tqdm(entries, desc="Preprocess HQColon"):
 58        out_path = os.path.join(preprocessed_dir, f"{entry['subject_id']}.h5")
 59        if os.path.exists(out_path):
 60            continue
 61
 62        series_dir = os.path.join(dicom_dir, entry["InstanceUID"])
 63        if not glob(os.path.join(series_dir, "*.dcm")):
 64            continue
 65
 66        gas_fluid_path = os.path.join(path, MASK_FOLDERS["gas_and_fluid"], entry["nnunet_label_file"])
 67        gas_path = os.path.join(path, MASK_FOLDERS["gas"], entry["nnunet_label_file"])
 68        if not (os.path.exists(gas_fluid_path) and os.path.exists(gas_path)):
 69            continue
 70
 71        volume, _ = util.load_dicom_series(series_dir)
 72        labels_gas_fluid = sitk.GetArrayFromImage(sitk.ReadImage(gas_fluid_path))
 73        labels_gas = sitk.GetArrayFromImage(sitk.ReadImage(gas_path))
 74
 75        assert volume.shape == labels_gas_fluid.shape == labels_gas.shape, \
 76            f"Shape mismatch for {entry['subject_id']}: {volume.shape}, {labels_gas_fluid.shape}, {labels_gas.shape}"
 77
 78        import h5py
 79        with h5py.File(out_path, "w") as f:
 80            f.create_dataset("raw", data=volume, compression="gzip")
 81            f.create_dataset("labels/gas_and_fluid", data=labels_gas_fluid.astype("uint8"), compression="gzip")
 82            f.create_dataset("labels/gas", data=labels_gas.astype("uint8"), compression="gzip")
 83
 84
 85def get_hqcolon_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 86    """Download the HQColon dataset.
 87
 88    Args:
 89        path: Filepath to a folder where the data is downloaded for further processing.
 90        download: Whether to download the data if it is not present.
 91
 92    Returns:
 93        Filepath where the preprocessed data is stored.
 94    """
 95    # NOTE: The preprocessing below skips volumes that were converted already, so an interrupted run resumes.
 96    preprocessed_dir = os.path.join(path, "preprocessed")
 97
 98    os.makedirs(path, exist_ok=True)
 99
100    metadata_path = os.path.join(path, "meta-data.json")
101    util.download_source(path=metadata_path, url=URLS["metadata"], download=download, checksum=CHECKSUMS["metadata"])
102    entries = _load_entries(metadata_path)
103
104    for name in ["gas_and_fluid", "gas"]:
105        mask_dir = os.path.join(path, MASK_FOLDERS[name])
106        if os.path.exists(mask_dir):
107            continue
108        zip_path = os.path.join(path, f"{name}.zip")
109        util.download_source(path=zip_path, url=URLS[name], download=download, checksum=CHECKSUMS[name])
110        util.unzip(zip_path=zip_path, dst=path)
111
112    dicom_dir = os.path.join(path, "dicom")
113    if download:
114        series_uids = [entry["InstanceUID"] for entry in entries]
115        util.download_tcia_series(series_uids, dst=dicom_dir, csv_filename=os.path.join(path, "hqcolon_series"))
116
117    _preprocess_hqcolon(path, entries, dicom_dir, preprocessed_dir)
118    return preprocessed_dir
119
120
121def get_hqcolon_paths(
122    path: Union[os.PathLike, str],
123    label_choice: Literal["gas_and_fluid", "gas"] = "gas_and_fluid",
124    download: bool = False,
125) -> List[str]:
126    """Get paths to the HQColon data.
127
128    Args:
129        path: Filepath to a folder where the data is downloaded for further processing.
130        label_choice: The choice of segmentation mask. Either 'gas_and_fluid' (the entire colon, including
131            collapsed segments and fluid) or 'gas' (only the gas-filled parts of the colon).
132        download: Whether to download the data if it is not present.
133
134    Returns:
135        List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data
136        ('labels/gas_and_fluid' and 'labels/gas').
137    """
138    if label_choice not in MASK_FOLDERS:
139        raise ValueError(f"'{label_choice}' is not a valid label choice. Choose from {list(MASK_FOLDERS.keys())}.")
140
141    preprocessed_dir = get_hqcolon_data(path, download)
142    volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5")))
143    assert len(volume_paths) > 0, f"Could not find any preprocessed samples in '{preprocessed_dir}'."
144    return volume_paths
145
146
147def get_hqcolon_dataset(
148    path: Union[os.PathLike, str],
149    patch_shape: Tuple[int, ...],
150    label_choice: Literal["gas_and_fluid", "gas"] = "gas_and_fluid",
151    resize_inputs: bool = False,
152    download: bool = False,
153    **kwargs
154) -> Dataset:
155    """Get the HQColon dataset for colon segmentation in CT colonography.
156
157    Args:
158        path: Filepath to a folder where the data is downloaded for further processing.
159        patch_shape: The patch shape to use for training.
160        label_choice: The choice of segmentation mask. Either 'gas_and_fluid' or 'gas'.
161        resize_inputs: Whether to resize inputs to the desired patch shape.
162        download: Whether to download the data if it is not present.
163        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
164
165    Returns:
166        The segmentation dataset.
167    """
168    volume_paths = get_hqcolon_paths(path, label_choice, download)
169
170    if resize_inputs:
171        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
172        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
173            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
174        )
175
176    return torch_em.default_segmentation_dataset(
177        raw_paths=volume_paths,
178        raw_key="raw",
179        label_paths=volume_paths,
180        label_key=f"labels/{label_choice}",
181        patch_shape=patch_shape,
182        is_seg_dataset=True,
183        **kwargs
184    )
185
186
187def get_hqcolon_loader(
188    path: Union[os.PathLike, str],
189    batch_size: int,
190    patch_shape: Tuple[int, ...],
191    label_choice: Literal["gas_and_fluid", "gas"] = "gas_and_fluid",
192    resize_inputs: bool = False,
193    download: bool = False,
194    **kwargs
195) -> DataLoader:
196    """Get the HQColon dataloader for colon segmentation in CT colonography.
197
198    Args:
199        path: Filepath to a folder where the data is downloaded for further processing.
200        batch_size: The batch size for training.
201        patch_shape: The patch shape to use for training.
202        label_choice: The choice of segmentation mask. Either 'gas_and_fluid' or 'gas'.
203        resize_inputs: Whether to resize inputs to the desired patch shape.
204        download: Whether to download the data if it is not present.
205        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
206
207    Returns:
208        The DataLoader.
209    """
210    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
211    dataset = get_hqcolon_dataset(path, patch_shape, label_choice, resize_inputs, download, **ds_kwargs)
212    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URLS = {'metadata': 'https://osf.io/download/8w6q7/', 'gas_and_fluid': 'https://osf.io/download/d4sc3/', 'gas': 'https://osf.io/download/y3ad2/'}
CHECKSUMS = {'metadata': '158bd6b4551c07f60ba3d32c7702ef67165b03308a5e1b5fa9e943598dd77693', 'gas_and_fluid': '99c0986b03291dbd0d4d973dc35bc5900e575fdb2f9ac9ea584381a5b12240bc', 'gas': '04bcb14aec9c4734756853f7c4b439b7de3c4a2ee30cedf482939a633cd4d840'}
MASK_FOLDERS = {'gas_and_fluid': 'Segmentation Air and Fluid', 'gas': 'Segmentation Air'}
def get_hqcolon_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 86def get_hqcolon_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 87    """Download the HQColon dataset.
 88
 89    Args:
 90        path: Filepath to a folder where the data is downloaded for further processing.
 91        download: Whether to download the data if it is not present.
 92
 93    Returns:
 94        Filepath where the preprocessed data is stored.
 95    """
 96    # NOTE: The preprocessing below skips volumes that were converted already, so an interrupted run resumes.
 97    preprocessed_dir = os.path.join(path, "preprocessed")
 98
 99    os.makedirs(path, exist_ok=True)
100
101    metadata_path = os.path.join(path, "meta-data.json")
102    util.download_source(path=metadata_path, url=URLS["metadata"], download=download, checksum=CHECKSUMS["metadata"])
103    entries = _load_entries(metadata_path)
104
105    for name in ["gas_and_fluid", "gas"]:
106        mask_dir = os.path.join(path, MASK_FOLDERS[name])
107        if os.path.exists(mask_dir):
108            continue
109        zip_path = os.path.join(path, f"{name}.zip")
110        util.download_source(path=zip_path, url=URLS[name], download=download, checksum=CHECKSUMS[name])
111        util.unzip(zip_path=zip_path, dst=path)
112
113    dicom_dir = os.path.join(path, "dicom")
114    if download:
115        series_uids = [entry["InstanceUID"] for entry in entries]
116        util.download_tcia_series(series_uids, dst=dicom_dir, csv_filename=os.path.join(path, "hqcolon_series"))
117
118    _preprocess_hqcolon(path, entries, dicom_dir, preprocessed_dir)
119    return preprocessed_dir

Download the HQColon dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the preprocessed data is stored.

def get_hqcolon_paths( path: Union[os.PathLike, str], label_choice: Literal['gas_and_fluid', 'gas'] = 'gas_and_fluid', download: bool = False) -> List[str]:
122def get_hqcolon_paths(
123    path: Union[os.PathLike, str],
124    label_choice: Literal["gas_and_fluid", "gas"] = "gas_and_fluid",
125    download: bool = False,
126) -> List[str]:
127    """Get paths to the HQColon data.
128
129    Args:
130        path: Filepath to a folder where the data is downloaded for further processing.
131        label_choice: The choice of segmentation mask. Either 'gas_and_fluid' (the entire colon, including
132            collapsed segments and fluid) or 'gas' (only the gas-filled parts of the colon).
133        download: Whether to download the data if it is not present.
134
135    Returns:
136        List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data
137        ('labels/gas_and_fluid' and 'labels/gas').
138    """
139    if label_choice not in MASK_FOLDERS:
140        raise ValueError(f"'{label_choice}' is not a valid label choice. Choose from {list(MASK_FOLDERS.keys())}.")
141
142    preprocessed_dir = get_hqcolon_data(path, download)
143    volume_paths = natsorted(glob(os.path.join(preprocessed_dir, "*.h5")))
144    assert len(volume_paths) > 0, f"Could not find any preprocessed samples in '{preprocessed_dir}'."
145    return volume_paths

Get paths to the HQColon data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • label_choice: The choice of segmentation mask. Either 'gas_and_fluid' (the entire colon, including collapsed segments and fluid) or 'gas' (only the gas-filled parts of the colon).
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the hdf5 files, which contain the image data ('raw') and the label data ('labels/gas_and_fluid' and 'labels/gas').

def get_hqcolon_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], label_choice: Literal['gas_and_fluid', 'gas'] = 'gas_and_fluid', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
148def get_hqcolon_dataset(
149    path: Union[os.PathLike, str],
150    patch_shape: Tuple[int, ...],
151    label_choice: Literal["gas_and_fluid", "gas"] = "gas_and_fluid",
152    resize_inputs: bool = False,
153    download: bool = False,
154    **kwargs
155) -> Dataset:
156    """Get the HQColon dataset for colon segmentation in CT colonography.
157
158    Args:
159        path: Filepath to a folder where the data is downloaded for further processing.
160        patch_shape: The patch shape to use for training.
161        label_choice: The choice of segmentation mask. Either 'gas_and_fluid' or 'gas'.
162        resize_inputs: Whether to resize inputs to the desired patch shape.
163        download: Whether to download the data if it is not present.
164        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
165
166    Returns:
167        The segmentation dataset.
168    """
169    volume_paths = get_hqcolon_paths(path, label_choice, download)
170
171    if resize_inputs:
172        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
173        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
174            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
175        )
176
177    return torch_em.default_segmentation_dataset(
178        raw_paths=volume_paths,
179        raw_key="raw",
180        label_paths=volume_paths,
181        label_key=f"labels/{label_choice}",
182        patch_shape=patch_shape,
183        is_seg_dataset=True,
184        **kwargs
185    )

Get the HQColon dataset for colon segmentation in CT colonography.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • label_choice: The choice of segmentation mask. Either 'gas_and_fluid' or 'gas'.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_hqcolon_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], label_choice: Literal['gas_and_fluid', 'gas'] = 'gas_and_fluid', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
188def get_hqcolon_loader(
189    path: Union[os.PathLike, str],
190    batch_size: int,
191    patch_shape: Tuple[int, ...],
192    label_choice: Literal["gas_and_fluid", "gas"] = "gas_and_fluid",
193    resize_inputs: bool = False,
194    download: bool = False,
195    **kwargs
196) -> DataLoader:
197    """Get the HQColon dataloader for colon segmentation in CT colonography.
198
199    Args:
200        path: Filepath to a folder where the data is downloaded for further processing.
201        batch_size: The batch size for training.
202        patch_shape: The patch shape to use for training.
203        label_choice: The choice of segmentation mask. Either 'gas_and_fluid' or 'gas'.
204        resize_inputs: Whether to resize inputs to the desired patch shape.
205        download: Whether to download the data if it is not present.
206        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
207
208    Returns:
209        The DataLoader.
210    """
211    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
212    dataset = get_hqcolon_dataset(path, patch_shape, label_choice, resize_inputs, download, **ds_kwargs)
213    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the HQColon dataloader for colon segmentation in CT colonography.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • label_choice: The choice of segmentation mask. Either 'gas_and_fluid' or 'gas'.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.