torch_em.data.datasets.medical.colonvessels

The ColonVessels 2026 dataset contains annotations for segmentation of mesenteric arteries and veins in dual-phase contrast-enhanced abdominal CT (CECT).

The dataset consists of 60 CECT studies (50 with both arterial and venous phases, 10 with the venous phase only) from adult patients, imaged on a Siemens SOMATOM Force scanner. Manual 3D segmentations of the mesenteric arteries (in the arterial phase) and veins (in the venous phase) were performed in 3D Slicer.

This dataset is from the publication https://doi.org/10.1038/s41597-026-07303-2. Please cite it if you use this dataset in your research.

  1"""The ColonVessels 2026 dataset contains annotations for segmentation of mesenteric arteries and veins
  2in dual-phase contrast-enhanced abdominal CT (CECT).
  3
  4The dataset consists of 60 CECT studies (50 with both arterial and venous phases, 10 with the venous phase
  5only) from adult patients, imaged on a Siemens SOMATOM Force scanner. Manual 3D segmentations of the
  6mesenteric arteries (in the arterial phase) and veins (in the venous phase) were performed in 3D Slicer.
  7
  8This dataset is from the publication https://doi.org/10.1038/s41597-026-07303-2.
  9Please cite it if you use this dataset in your research.
 10"""
 11
 12import os
 13from glob import glob
 14from pathlib import Path
 15from natsort import natsorted
 16from typing import Union, Tuple, List, Literal, Optional
 17
 18from torch.utils.data import Dataset, DataLoader
 19
 20import torch_em
 21
 22from .. import util
 23
 24
 25URL = "https://zenodo.org/records/17407158/files/data.zip"
 26CHECKSUM = "a5ae3148a54805165388b648a1a45c793ec535ef4eea3b68c679704b3c6db7b5"
 27
 28VESSEL_TYPES = ["arteries", "veins"]
 29
 30
 31def _convert_image_to_nifti(nrrd_path: str, nifti_path: str) -> None:
 32    if os.path.exists(nifti_path):
 33        return
 34
 35    import SimpleITK as sitk
 36
 37    image = sitk.ReadImage(nrrd_path)
 38    sitk.WriteImage(image, nifti_path, useCompression=True)
 39
 40
 41def _convert_mask_to_nifti(nrrd_path: str, nifti_path: str) -> None:
 42    """Convert a 3D Slicer '.seg.nrrd' segmentation to a single-channel binary foreground mask.
 43
 44    The segmentations combine multiple named vessel segments (e.g. individual named arteries) into a small
 45    number of shared binary labelmap layers (a 4th array axis), to avoid one layer per segment. Since this
 46    dataset is used for binary vessel (foreground) segmentation, the layers are collapsed into one channel
 47    by marking a voxel as foreground if it is non-zero in any layer.
 48    """
 49    if os.path.exists(nifti_path):
 50        return
 51
 52    import SimpleITK as sitk
 53
 54    seg = sitk.ReadImage(nrrd_path)
 55    arr = sitk.GetArrayFromImage(seg)
 56    if arr.ndim == 4:  # (Z, Y, X, num_layers) -> collapse the layers into one binary foreground mask.
 57        arr = (arr > 0).any(axis=-1)
 58    mask = (arr > 0).astype("uint8")
 59
 60    mask_image = sitk.GetImageFromArray(mask)
 61    mask_image.SetSpacing(seg.GetSpacing()[:3])
 62    mask_image.SetOrigin(seg.GetOrigin()[:3])
 63    sitk.WriteImage(mask_image, nifti_path, useCompression=True)
 64
 65
 66def get_colonvessels_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 67    """Download the ColonVessels 2026 dataset.
 68
 69    Args:
 70        path: Filepath to a folder where the data is downloaded for further processing.
 71        download: Whether to download the data if it is not present.
 72
 73    Returns:
 74        Filepath where the data is stored.
 75    """
 76    data_dir = os.path.join(path, "data")
 77    if os.path.exists(data_dir):
 78        return data_dir
 79
 80    os.makedirs(path, exist_ok=True)
 81
 82    zip_path = os.path.join(path, "data.zip")
 83    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 84    util.unzip(zip_path=zip_path, dst=path)
 85
 86    return data_dir
 87
 88
 89def get_colonvessels_paths(
 90    path: Union[os.PathLike, str],
 91    vessel_type: Optional[Literal["arteries", "veins"]] = None,
 92    download: bool = False,
 93) -> Tuple[List[str], List[str]]:
 94    """Get paths to the ColonVessels 2026 data.
 95
 96    Args:
 97        path: Filepath to a folder where the data is downloaded for further processing.
 98        vessel_type: The choice of vessel type to segment.
 99        download: Whether to download the data if it is not present.
100
101    Returns:
102        List of filepaths for the image data.
103        List of filepaths for the label data.
104    """
105    data_dir = get_colonvessels_data(path, download)
106
107    if vessel_type is None:
108        vessel_types = VESSEL_TYPES
109    else:
110        assert vessel_type in VESSEL_TYPES, f"'{vessel_type}' is not a valid vessel type."
111        vessel_types = [vessel_type]
112
113    phase_per_vessel_type = {"arteries": "Arterial", "veins": "Venous"}
114
115    nifti_dir = os.path.join(path, "nifti")
116    os.makedirs(nifti_dir, exist_ok=True)
117
118    image_paths, gt_paths = [], []
119    for patient_dir in natsorted(glob(os.path.join(data_dir, "pat_*"))):
120        patient_id = os.path.basename(patient_dir)
121        for _vessel_type in vessel_types:
122            phase = phase_per_vessel_type[_vessel_type]
123            label_name = _vessel_type.capitalize()
124            image_nrrd = os.path.join(patient_dir, f"{patient_id}_{phase}_Phase_CT.nrrd")
125            gt_nrrd = os.path.join(patient_dir, f"{patient_id}_{phase}_Phase_{label_name}.seg.nrrd")
126            if not (os.path.exists(image_nrrd) and os.path.exists(gt_nrrd)):
127                continue
128
129            image_path = os.path.join(nifti_dir, f"{Path(image_nrrd).stem}.nii.gz")
130            gt_path = os.path.join(nifti_dir, f"{Path(gt_nrrd).stem.replace('.seg', '')}.nii.gz")
131
132            _convert_image_to_nifti(image_nrrd, image_path)
133            _convert_mask_to_nifti(gt_nrrd, gt_path)
134
135            image_paths.append(image_path)
136            gt_paths.append(gt_path)
137
138    assert len(image_paths) > 0 and len(image_paths) == len(gt_paths), \
139        f"Could not find a matching number of images and labels in '{data_dir}'."
140
141    return image_paths, gt_paths
142
143
144def get_colonvessels_dataset(
145    path: Union[os.PathLike, str],
146    patch_shape: Tuple[int, ...],
147    vessel_type: Optional[Literal["arteries", "veins"]] = None,
148    resize_inputs: bool = False,
149    download: bool = False,
150    **kwargs
151) -> Dataset:
152    """Get the ColonVessels 2026 dataset for segmentation of mesenteric arteries and veins.
153
154    Args:
155        path: Filepath to a folder where the data is downloaded for further processing.
156        patch_shape: The patch shape to use for training.
157        vessel_type: The choice of vessel type to segment.
158        resize_inputs: Whether to resize the inputs to the patch shape.
159        download: Whether to download the data if it is not present.
160        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
161
162    Returns:
163        The segmentation dataset.
164    """
165    image_paths, gt_paths = get_colonvessels_paths(path, vessel_type, download)
166
167    if resize_inputs:
168        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
169        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
170            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
171        )
172
173    return torch_em.default_segmentation_dataset(
174        raw_paths=image_paths,
175        raw_key="data",
176        label_paths=gt_paths,
177        label_key="data",
178        patch_shape=patch_shape,
179        is_seg_dataset=True,
180        **kwargs
181    )
182
183
184def get_colonvessels_loader(
185    path: Union[os.PathLike, str],
186    batch_size: int,
187    patch_shape: Tuple[int, ...],
188    vessel_type: Optional[Literal["arteries", "veins"]] = None,
189    resize_inputs: bool = False,
190    download: bool = False,
191    **kwargs
192) -> DataLoader:
193    """Get the ColonVessels 2026 dataloader for segmentation of mesenteric arteries and veins.
194
195    Args:
196        path: Filepath to a folder where the data is downloaded for further processing.
197        batch_size: The batch size for training.
198        patch_shape: The patch shape to use for training.
199        vessel_type: The choice of vessel type to segment.
200        resize_inputs: Whether to resize the inputs to the patch shape.
201        download: Whether to download the data if it is not present.
202        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or the PyTorch DataLoader.
203
204    Returns:
205        The DataLoader.
206    """
207    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
208    dataset = get_colonvessels_dataset(path, patch_shape, vessel_type, resize_inputs, download, **ds_kwargs)
209    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/records/17407158/files/data.zip'
CHECKSUM = 'a5ae3148a54805165388b648a1a45c793ec535ef4eea3b68c679704b3c6db7b5'
VESSEL_TYPES = ['arteries', 'veins']
def get_colonvessels_data(path: Union[os.PathLike, str], download: bool = False) -> str:
67def get_colonvessels_data(path: Union[os.PathLike, str], download: bool = False) -> str:
68    """Download the ColonVessels 2026 dataset.
69
70    Args:
71        path: Filepath to a folder where the data is downloaded for further processing.
72        download: Whether to download the data if it is not present.
73
74    Returns:
75        Filepath where the data is stored.
76    """
77    data_dir = os.path.join(path, "data")
78    if os.path.exists(data_dir):
79        return data_dir
80
81    os.makedirs(path, exist_ok=True)
82
83    zip_path = os.path.join(path, "data.zip")
84    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
85    util.unzip(zip_path=zip_path, dst=path)
86
87    return data_dir

Download the ColonVessels 2026 dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is stored.

def get_colonvessels_paths( path: Union[os.PathLike, str], vessel_type: Optional[Literal['arteries', 'veins']] = None, download: bool = False) -> Tuple[List[str], List[str]]:
 90def get_colonvessels_paths(
 91    path: Union[os.PathLike, str],
 92    vessel_type: Optional[Literal["arteries", "veins"]] = None,
 93    download: bool = False,
 94) -> Tuple[List[str], List[str]]:
 95    """Get paths to the ColonVessels 2026 data.
 96
 97    Args:
 98        path: Filepath to a folder where the data is downloaded for further processing.
 99        vessel_type: The choice of vessel type to segment.
100        download: Whether to download the data if it is not present.
101
102    Returns:
103        List of filepaths for the image data.
104        List of filepaths for the label data.
105    """
106    data_dir = get_colonvessels_data(path, download)
107
108    if vessel_type is None:
109        vessel_types = VESSEL_TYPES
110    else:
111        assert vessel_type in VESSEL_TYPES, f"'{vessel_type}' is not a valid vessel type."
112        vessel_types = [vessel_type]
113
114    phase_per_vessel_type = {"arteries": "Arterial", "veins": "Venous"}
115
116    nifti_dir = os.path.join(path, "nifti")
117    os.makedirs(nifti_dir, exist_ok=True)
118
119    image_paths, gt_paths = [], []
120    for patient_dir in natsorted(glob(os.path.join(data_dir, "pat_*"))):
121        patient_id = os.path.basename(patient_dir)
122        for _vessel_type in vessel_types:
123            phase = phase_per_vessel_type[_vessel_type]
124            label_name = _vessel_type.capitalize()
125            image_nrrd = os.path.join(patient_dir, f"{patient_id}_{phase}_Phase_CT.nrrd")
126            gt_nrrd = os.path.join(patient_dir, f"{patient_id}_{phase}_Phase_{label_name}.seg.nrrd")
127            if not (os.path.exists(image_nrrd) and os.path.exists(gt_nrrd)):
128                continue
129
130            image_path = os.path.join(nifti_dir, f"{Path(image_nrrd).stem}.nii.gz")
131            gt_path = os.path.join(nifti_dir, f"{Path(gt_nrrd).stem.replace('.seg', '')}.nii.gz")
132
133            _convert_image_to_nifti(image_nrrd, image_path)
134            _convert_mask_to_nifti(gt_nrrd, gt_path)
135
136            image_paths.append(image_path)
137            gt_paths.append(gt_path)
138
139    assert len(image_paths) > 0 and len(image_paths) == len(gt_paths), \
140        f"Could not find a matching number of images and labels in '{data_dir}'."
141
142    return image_paths, gt_paths

Get paths to the ColonVessels 2026 data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • vessel_type: The choice of vessel type to segment.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_colonvessels_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, ...], vessel_type: Optional[Literal['arteries', 'veins']] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
145def get_colonvessels_dataset(
146    path: Union[os.PathLike, str],
147    patch_shape: Tuple[int, ...],
148    vessel_type: Optional[Literal["arteries", "veins"]] = None,
149    resize_inputs: bool = False,
150    download: bool = False,
151    **kwargs
152) -> Dataset:
153    """Get the ColonVessels 2026 dataset for segmentation of mesenteric arteries and veins.
154
155    Args:
156        path: Filepath to a folder where the data is downloaded for further processing.
157        patch_shape: The patch shape to use for training.
158        vessel_type: The choice of vessel type to segment.
159        resize_inputs: Whether to resize the inputs to the patch shape.
160        download: Whether to download the data if it is not present.
161        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
162
163    Returns:
164        The segmentation dataset.
165    """
166    image_paths, gt_paths = get_colonvessels_paths(path, vessel_type, download)
167
168    if resize_inputs:
169        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
170        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
171            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
172        )
173
174    return torch_em.default_segmentation_dataset(
175        raw_paths=image_paths,
176        raw_key="data",
177        label_paths=gt_paths,
178        label_key="data",
179        patch_shape=patch_shape,
180        is_seg_dataset=True,
181        **kwargs
182    )

Get the ColonVessels 2026 dataset for segmentation of mesenteric arteries and veins.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • vessel_type: The choice of vessel type to segment.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_colonvessels_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, ...], vessel_type: Optional[Literal['arteries', 'veins']] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
185def get_colonvessels_loader(
186    path: Union[os.PathLike, str],
187    batch_size: int,
188    patch_shape: Tuple[int, ...],
189    vessel_type: Optional[Literal["arteries", "veins"]] = None,
190    resize_inputs: bool = False,
191    download: bool = False,
192    **kwargs
193) -> DataLoader:
194    """Get the ColonVessels 2026 dataloader for segmentation of mesenteric arteries and veins.
195
196    Args:
197        path: Filepath to a folder where the data is downloaded for further processing.
198        batch_size: The batch size for training.
199        patch_shape: The patch shape to use for training.
200        vessel_type: The choice of vessel type to segment.
201        resize_inputs: Whether to resize the inputs to the patch shape.
202        download: Whether to download the data if it is not present.
203        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or the PyTorch DataLoader.
204
205    Returns:
206        The DataLoader.
207    """
208    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
209    dataset = get_colonvessels_dataset(path, patch_shape, vessel_type, resize_inputs, download, **ds_kwargs)
210    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the ColonVessels 2026 dataloader for segmentation of mesenteric arteries and veins.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • vessel_type: The choice of vessel type to segment.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or the PyTorch DataLoader.
Returns:

The DataLoader.