torch_em.data.datasets.medical.colonvessels
The ColonVessels 2026 dataset contains annotations for segmentation of mesenteric arteries and veins in dual-phase contrast-enhanced abdominal CT (CECT).
The dataset consists of 60 CECT studies (50 with both arterial and venous phases, 10 with the venous phase only) from adult patients, imaged on a Siemens SOMATOM Force scanner. Manual 3D segmentations of the mesenteric arteries (in the arterial phase) and veins (in the venous phase) were performed in 3D Slicer.
This dataset is from the publication https://doi.org/10.1038/s41597-026-07303-2. Please cite it if you use this dataset in your research.
1"""The ColonVessels 2026 dataset contains annotations for segmentation of mesenteric arteries and veins 2in dual-phase contrast-enhanced abdominal CT (CECT). 3 4The dataset consists of 60 CECT studies (50 with both arterial and venous phases, 10 with the venous phase 5only) from adult patients, imaged on a Siemens SOMATOM Force scanner. Manual 3D segmentations of the 6mesenteric arteries (in the arterial phase) and veins (in the venous phase) were performed in 3D Slicer. 7 8This dataset is from the publication https://doi.org/10.1038/s41597-026-07303-2. 9Please cite it if you use this dataset in your research. 10""" 11 12import os 13from glob import glob 14from pathlib import Path 15from natsort import natsorted 16from typing import Union, Tuple, List, Literal, Optional 17 18from torch.utils.data import Dataset, DataLoader 19 20import torch_em 21 22from .. import util 23 24 25URL = "https://zenodo.org/records/17407158/files/data.zip" 26CHECKSUM = "a5ae3148a54805165388b648a1a45c793ec535ef4eea3b68c679704b3c6db7b5" 27 28VESSEL_TYPES = ["arteries", "veins"] 29 30 31def _convert_image_to_nifti(nrrd_path: str, nifti_path: str) -> None: 32 if os.path.exists(nifti_path): 33 return 34 35 import SimpleITK as sitk 36 37 image = sitk.ReadImage(nrrd_path) 38 sitk.WriteImage(image, nifti_path, useCompression=True) 39 40 41def _convert_mask_to_nifti(nrrd_path: str, nifti_path: str) -> None: 42 """Convert a 3D Slicer '.seg.nrrd' segmentation to a single-channel binary foreground mask. 43 44 The segmentations combine multiple named vessel segments (e.g. individual named arteries) into a small 45 number of shared binary labelmap layers (a 4th array axis), to avoid one layer per segment. Since this 46 dataset is used for binary vessel (foreground) segmentation, the layers are collapsed into one channel 47 by marking a voxel as foreground if it is non-zero in any layer. 48 """ 49 if os.path.exists(nifti_path): 50 return 51 52 import SimpleITK as sitk 53 54 seg = sitk.ReadImage(nrrd_path) 55 arr = sitk.GetArrayFromImage(seg) 56 if arr.ndim == 4: # (Z, Y, X, num_layers) -> collapse the layers into one binary foreground mask. 57 arr = (arr > 0).any(axis=-1) 58 mask = (arr > 0).astype("uint8") 59 60 mask_image = sitk.GetImageFromArray(mask) 61 mask_image.SetSpacing(seg.GetSpacing()[:3]) 62 mask_image.SetOrigin(seg.GetOrigin()[:3]) 63 sitk.WriteImage(mask_image, nifti_path, useCompression=True) 64 65 66def get_colonvessels_data(path: Union[os.PathLike, str], download: bool = False) -> str: 67 """Download the ColonVessels 2026 dataset. 68 69 Args: 70 path: Filepath to a folder where the data is downloaded for further processing. 71 download: Whether to download the data if it is not present. 72 73 Returns: 74 Filepath where the data is stored. 75 """ 76 data_dir = os.path.join(path, "data") 77 if os.path.exists(data_dir): 78 return data_dir 79 80 os.makedirs(path, exist_ok=True) 81 82 zip_path = os.path.join(path, "data.zip") 83 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 84 util.unzip(zip_path=zip_path, dst=path) 85 86 return data_dir 87 88 89def get_colonvessels_paths( 90 path: Union[os.PathLike, str], 91 vessel_type: Optional[Literal["arteries", "veins"]] = None, 92 download: bool = False, 93) -> Tuple[List[str], List[str]]: 94 """Get paths to the ColonVessels 2026 data. 95 96 Args: 97 path: Filepath to a folder where the data is downloaded for further processing. 98 vessel_type: The choice of vessel type to segment. 99 download: Whether to download the data if it is not present. 100 101 Returns: 102 List of filepaths for the image data. 103 List of filepaths for the label data. 104 """ 105 data_dir = get_colonvessels_data(path, download) 106 107 if vessel_type is None: 108 vessel_types = VESSEL_TYPES 109 else: 110 assert vessel_type in VESSEL_TYPES, f"'{vessel_type}' is not a valid vessel type." 111 vessel_types = [vessel_type] 112 113 phase_per_vessel_type = {"arteries": "Arterial", "veins": "Venous"} 114 115 nifti_dir = os.path.join(path, "nifti") 116 os.makedirs(nifti_dir, exist_ok=True) 117 118 image_paths, gt_paths = [], [] 119 for patient_dir in natsorted(glob(os.path.join(data_dir, "pat_*"))): 120 patient_id = os.path.basename(patient_dir) 121 for _vessel_type in vessel_types: 122 phase = phase_per_vessel_type[_vessel_type] 123 label_name = _vessel_type.capitalize() 124 image_nrrd = os.path.join(patient_dir, f"{patient_id}_{phase}_Phase_CT.nrrd") 125 gt_nrrd = os.path.join(patient_dir, f"{patient_id}_{phase}_Phase_{label_name}.seg.nrrd") 126 if not (os.path.exists(image_nrrd) and os.path.exists(gt_nrrd)): 127 continue 128 129 image_path = os.path.join(nifti_dir, f"{Path(image_nrrd).stem}.nii.gz") 130 gt_path = os.path.join(nifti_dir, f"{Path(gt_nrrd).stem.replace('.seg', '')}.nii.gz") 131 132 _convert_image_to_nifti(image_nrrd, image_path) 133 _convert_mask_to_nifti(gt_nrrd, gt_path) 134 135 image_paths.append(image_path) 136 gt_paths.append(gt_path) 137 138 assert len(image_paths) > 0 and len(image_paths) == len(gt_paths), \ 139 f"Could not find a matching number of images and labels in '{data_dir}'." 140 141 return image_paths, gt_paths 142 143 144def get_colonvessels_dataset( 145 path: Union[os.PathLike, str], 146 patch_shape: Tuple[int, ...], 147 vessel_type: Optional[Literal["arteries", "veins"]] = None, 148 resize_inputs: bool = False, 149 download: bool = False, 150 **kwargs 151) -> Dataset: 152 """Get the ColonVessels 2026 dataset for segmentation of mesenteric arteries and veins. 153 154 Args: 155 path: Filepath to a folder where the data is downloaded for further processing. 156 patch_shape: The patch shape to use for training. 157 vessel_type: The choice of vessel type to segment. 158 resize_inputs: Whether to resize the inputs to the patch shape. 159 download: Whether to download the data if it is not present. 160 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 161 162 Returns: 163 The segmentation dataset. 164 """ 165 image_paths, gt_paths = get_colonvessels_paths(path, vessel_type, download) 166 167 if resize_inputs: 168 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 169 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 170 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 171 ) 172 173 return torch_em.default_segmentation_dataset( 174 raw_paths=image_paths, 175 raw_key="data", 176 label_paths=gt_paths, 177 label_key="data", 178 patch_shape=patch_shape, 179 is_seg_dataset=True, 180 **kwargs 181 ) 182 183 184def get_colonvessels_loader( 185 path: Union[os.PathLike, str], 186 batch_size: int, 187 patch_shape: Tuple[int, ...], 188 vessel_type: Optional[Literal["arteries", "veins"]] = None, 189 resize_inputs: bool = False, 190 download: bool = False, 191 **kwargs 192) -> DataLoader: 193 """Get the ColonVessels 2026 dataloader for segmentation of mesenteric arteries and veins. 194 195 Args: 196 path: Filepath to a folder where the data is downloaded for further processing. 197 batch_size: The batch size for training. 198 patch_shape: The patch shape to use for training. 199 vessel_type: The choice of vessel type to segment. 200 resize_inputs: Whether to resize the inputs to the patch shape. 201 download: Whether to download the data if it is not present. 202 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or the PyTorch DataLoader. 203 204 Returns: 205 The DataLoader. 206 """ 207 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 208 dataset = get_colonvessels_dataset(path, patch_shape, vessel_type, resize_inputs, download, **ds_kwargs) 209 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
67def get_colonvessels_data(path: Union[os.PathLike, str], download: bool = False) -> str: 68 """Download the ColonVessels 2026 dataset. 69 70 Args: 71 path: Filepath to a folder where the data is downloaded for further processing. 72 download: Whether to download the data if it is not present. 73 74 Returns: 75 Filepath where the data is stored. 76 """ 77 data_dir = os.path.join(path, "data") 78 if os.path.exists(data_dir): 79 return data_dir 80 81 os.makedirs(path, exist_ok=True) 82 83 zip_path = os.path.join(path, "data.zip") 84 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 85 util.unzip(zip_path=zip_path, dst=path) 86 87 return data_dir
Download the ColonVessels 2026 dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is stored.
90def get_colonvessels_paths( 91 path: Union[os.PathLike, str], 92 vessel_type: Optional[Literal["arteries", "veins"]] = None, 93 download: bool = False, 94) -> Tuple[List[str], List[str]]: 95 """Get paths to the ColonVessels 2026 data. 96 97 Args: 98 path: Filepath to a folder where the data is downloaded for further processing. 99 vessel_type: The choice of vessel type to segment. 100 download: Whether to download the data if it is not present. 101 102 Returns: 103 List of filepaths for the image data. 104 List of filepaths for the label data. 105 """ 106 data_dir = get_colonvessels_data(path, download) 107 108 if vessel_type is None: 109 vessel_types = VESSEL_TYPES 110 else: 111 assert vessel_type in VESSEL_TYPES, f"'{vessel_type}' is not a valid vessel type." 112 vessel_types = [vessel_type] 113 114 phase_per_vessel_type = {"arteries": "Arterial", "veins": "Venous"} 115 116 nifti_dir = os.path.join(path, "nifti") 117 os.makedirs(nifti_dir, exist_ok=True) 118 119 image_paths, gt_paths = [], [] 120 for patient_dir in natsorted(glob(os.path.join(data_dir, "pat_*"))): 121 patient_id = os.path.basename(patient_dir) 122 for _vessel_type in vessel_types: 123 phase = phase_per_vessel_type[_vessel_type] 124 label_name = _vessel_type.capitalize() 125 image_nrrd = os.path.join(patient_dir, f"{patient_id}_{phase}_Phase_CT.nrrd") 126 gt_nrrd = os.path.join(patient_dir, f"{patient_id}_{phase}_Phase_{label_name}.seg.nrrd") 127 if not (os.path.exists(image_nrrd) and os.path.exists(gt_nrrd)): 128 continue 129 130 image_path = os.path.join(nifti_dir, f"{Path(image_nrrd).stem}.nii.gz") 131 gt_path = os.path.join(nifti_dir, f"{Path(gt_nrrd).stem.replace('.seg', '')}.nii.gz") 132 133 _convert_image_to_nifti(image_nrrd, image_path) 134 _convert_mask_to_nifti(gt_nrrd, gt_path) 135 136 image_paths.append(image_path) 137 gt_paths.append(gt_path) 138 139 assert len(image_paths) > 0 and len(image_paths) == len(gt_paths), \ 140 f"Could not find a matching number of images and labels in '{data_dir}'." 141 142 return image_paths, gt_paths
Get paths to the ColonVessels 2026 data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- vessel_type: The choice of vessel type to segment.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
145def get_colonvessels_dataset( 146 path: Union[os.PathLike, str], 147 patch_shape: Tuple[int, ...], 148 vessel_type: Optional[Literal["arteries", "veins"]] = None, 149 resize_inputs: bool = False, 150 download: bool = False, 151 **kwargs 152) -> Dataset: 153 """Get the ColonVessels 2026 dataset for segmentation of mesenteric arteries and veins. 154 155 Args: 156 path: Filepath to a folder where the data is downloaded for further processing. 157 patch_shape: The patch shape to use for training. 158 vessel_type: The choice of vessel type to segment. 159 resize_inputs: Whether to resize the inputs to the patch shape. 160 download: Whether to download the data if it is not present. 161 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 162 163 Returns: 164 The segmentation dataset. 165 """ 166 image_paths, gt_paths = get_colonvessels_paths(path, vessel_type, download) 167 168 if resize_inputs: 169 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 170 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 171 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 172 ) 173 174 return torch_em.default_segmentation_dataset( 175 raw_paths=image_paths, 176 raw_key="data", 177 label_paths=gt_paths, 178 label_key="data", 179 patch_shape=patch_shape, 180 is_seg_dataset=True, 181 **kwargs 182 )
Get the ColonVessels 2026 dataset for segmentation of mesenteric arteries and veins.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- vessel_type: The choice of vessel type to segment.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
185def get_colonvessels_loader( 186 path: Union[os.PathLike, str], 187 batch_size: int, 188 patch_shape: Tuple[int, ...], 189 vessel_type: Optional[Literal["arteries", "veins"]] = None, 190 resize_inputs: bool = False, 191 download: bool = False, 192 **kwargs 193) -> DataLoader: 194 """Get the ColonVessels 2026 dataloader for segmentation of mesenteric arteries and veins. 195 196 Args: 197 path: Filepath to a folder where the data is downloaded for further processing. 198 batch_size: The batch size for training. 199 patch_shape: The patch shape to use for training. 200 vessel_type: The choice of vessel type to segment. 201 resize_inputs: Whether to resize the inputs to the patch shape. 202 download: Whether to download the data if it is not present. 203 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or the PyTorch DataLoader. 204 205 Returns: 206 The DataLoader. 207 """ 208 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 209 dataset = get_colonvessels_dataset(path, patch_shape, vessel_type, resize_inputs, download, **ds_kwargs) 210 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the ColonVessels 2026 dataloader for segmentation of mesenteric arteries and veins.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- vessel_type: The choice of vessel type to segment.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor the PyTorch DataLoader.
Returns:
The DataLoader.