torch_em.data.datasets.medical.cvc_colondb

The CVC-ColonDB dataset contains annotations for polyp segmentation in colonoscopy images.

The dataset consists of 380 still colonoscopy frames extracted from 15 different video sequences, each paired with a binary segmentation mask of the polyp region.

The dataset is located at https://www.kaggle.com/datasets/longvil/cvc-colondb. This is a mirror of the original CVC-ColonDB release from the Computer Vision Center (CVC), Barcelona, which is gated behind manual registration on the CVC-Colon website (https://pages.cvc.uab.es/CVC-Colon/).

This dataset is from the publication https://doi.org/10.1016/j.patcog.2012.03.002. Please cite it if you use this dataset for your research.

  1"""The CVC-ColonDB dataset contains annotations for polyp segmentation in colonoscopy images.
  2
  3The dataset consists of 380 still colonoscopy frames extracted from 15 different video sequences,
  4each paired with a binary segmentation mask of the polyp region.
  5
  6The dataset is located at https://www.kaggle.com/datasets/longvil/cvc-colondb. This is a mirror
  7of the original CVC-ColonDB release from the Computer Vision Center (CVC), Barcelona, which is
  8gated behind manual registration on the CVC-Colon website (https://pages.cvc.uab.es/CVC-Colon/).
  9
 10This dataset is from the publication https://doi.org/10.1016/j.patcog.2012.03.002.
 11Please cite it if you use this dataset for your research.
 12"""
 13
 14import os
 15from glob import glob
 16from natsort import natsorted
 17from typing import Union, Tuple, List
 18
 19from torch.utils.data import Dataset, DataLoader
 20
 21import torch_em
 22
 23from .. import util
 24
 25
 26KAGGLE_DATASET_NAME = "longvil/cvc-colondb"
 27
 28
 29def get_cvc_colondb_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 30    """Download the CVC-ColonDB dataset.
 31
 32    Args:
 33        path: Filepath to a folder where the data is downloaded for further processing.
 34        download: Whether to download the data if it is not present.
 35
 36    Returns:
 37        Filepath where the data is downloaded.
 38    """
 39    data_dir = os.path.join(path, "CVC-ColonDB")
 40    if os.path.exists(data_dir):
 41        return data_dir
 42
 43    os.makedirs(path, exist_ok=True)
 44
 45    util.download_source_kaggle(path=path, dataset_name=KAGGLE_DATASET_NAME, download=download)
 46
 47    zip_path = os.path.join(path, "cvc-colondb.zip")
 48    util.unzip(zip_path=zip_path, dst=path)
 49
 50    if not os.path.exists(data_dir):
 51        raise RuntimeError(f"The dataset could not be found at '{path}' after extraction.")
 52
 53    return data_dir
 54
 55
 56def get_cvc_colondb_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 57    """Get paths to the CVC-ColonDB data.
 58
 59    Args:
 60        path: Filepath to a folder where the data is downloaded for further processing.
 61        download: Whether to download the data if it is not present.
 62
 63    Returns:
 64        List of filepaths for the image data.
 65        List of filepaths for the label data.
 66    """
 67    data_dir = get_cvc_colondb_data(path=path, download=download)
 68
 69    image_paths = natsorted(glob(os.path.join(data_dir, "images", "*.png")))
 70    gt_paths = natsorted(glob(os.path.join(data_dir, "masks", "*.png")))
 71
 72    if len(image_paths) == 0 or len(image_paths) != len(gt_paths):
 73        raise RuntimeError("Something went wrong with fetching the image and label paths.")
 74
 75    return image_paths, gt_paths
 76
 77
 78def get_cvc_colondb_dataset(
 79    path: Union[os.PathLike, str],
 80    patch_shape: Tuple[int, int],
 81    resize_inputs: bool = False,
 82    download: bool = False,
 83    **kwargs
 84) -> Dataset:
 85    """Get the CVC-ColonDB dataset for polyp segmentation.
 86
 87    Args:
 88        path: Filepath to a folder where the data is downloaded for further processing.
 89        patch_shape: The patch shape to use for training.
 90        resize_inputs: Whether to resize the inputs to the patch shape.
 91        download: Whether to download the data if it is not present.
 92        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
 93
 94    Returns:
 95        The segmentation dataset.
 96    """
 97    image_paths, gt_paths = get_cvc_colondb_paths(path, download)
 98
 99    if resize_inputs:
100        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
101        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
102            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
103        )
104
105    return torch_em.default_segmentation_dataset(
106        raw_paths=image_paths,
107        raw_key=None,
108        label_paths=gt_paths,
109        label_key=None,
110        patch_shape=patch_shape,
111        is_seg_dataset=False,
112        **kwargs
113    )
114
115
116def get_cvc_colondb_loader(
117    path: Union[os.PathLike, str],
118    patch_shape: Tuple[int, int],
119    batch_size: int,
120    resize_inputs: bool = False,
121    download: bool = False,
122    **kwargs
123) -> DataLoader:
124    """Get the CVC-ColonDB dataloader for polyp segmentation.
125
126    Args:
127        path: Filepath to a folder where the data is downloaded for further processing.
128        patch_shape: The patch shape to use for training.
129        batch_size: The batch size for training.
130        resize_inputs: Whether to resize the inputs to the patch shape.
131        download: Whether to download the data if it is not present.
132        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
133
134    Returns:
135        The DataLoader.
136    """
137    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
138    dataset = get_cvc_colondb_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
139    return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
KAGGLE_DATASET_NAME = 'longvil/cvc-colondb'
def get_cvc_colondb_data(path: Union[os.PathLike, str], download: bool = False) -> str:
30def get_cvc_colondb_data(path: Union[os.PathLike, str], download: bool = False) -> str:
31    """Download the CVC-ColonDB dataset.
32
33    Args:
34        path: Filepath to a folder where the data is downloaded for further processing.
35        download: Whether to download the data if it is not present.
36
37    Returns:
38        Filepath where the data is downloaded.
39    """
40    data_dir = os.path.join(path, "CVC-ColonDB")
41    if os.path.exists(data_dir):
42        return data_dir
43
44    os.makedirs(path, exist_ok=True)
45
46    util.download_source_kaggle(path=path, dataset_name=KAGGLE_DATASET_NAME, download=download)
47
48    zip_path = os.path.join(path, "cvc-colondb.zip")
49    util.unzip(zip_path=zip_path, dst=path)
50
51    if not os.path.exists(data_dir):
52        raise RuntimeError(f"The dataset could not be found at '{path}' after extraction.")
53
54    return data_dir

Download the CVC-ColonDB dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_cvc_colondb_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
57def get_cvc_colondb_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
58    """Get paths to the CVC-ColonDB data.
59
60    Args:
61        path: Filepath to a folder where the data is downloaded for further processing.
62        download: Whether to download the data if it is not present.
63
64    Returns:
65        List of filepaths for the image data.
66        List of filepaths for the label data.
67    """
68    data_dir = get_cvc_colondb_data(path=path, download=download)
69
70    image_paths = natsorted(glob(os.path.join(data_dir, "images", "*.png")))
71    gt_paths = natsorted(glob(os.path.join(data_dir, "masks", "*.png")))
72
73    if len(image_paths) == 0 or len(image_paths) != len(gt_paths):
74        raise RuntimeError("Something went wrong with fetching the image and label paths.")
75
76    return image_paths, gt_paths

Get paths to the CVC-ColonDB data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_cvc_colondb_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
 79def get_cvc_colondb_dataset(
 80    path: Union[os.PathLike, str],
 81    patch_shape: Tuple[int, int],
 82    resize_inputs: bool = False,
 83    download: bool = False,
 84    **kwargs
 85) -> Dataset:
 86    """Get the CVC-ColonDB dataset for polyp segmentation.
 87
 88    Args:
 89        path: Filepath to a folder where the data is downloaded for further processing.
 90        patch_shape: The patch shape to use for training.
 91        resize_inputs: Whether to resize the inputs to the patch shape.
 92        download: Whether to download the data if it is not present.
 93        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
 94
 95    Returns:
 96        The segmentation dataset.
 97    """
 98    image_paths, gt_paths = get_cvc_colondb_paths(path, download)
 99
100    if resize_inputs:
101        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
102        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
103            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
104        )
105
106    return torch_em.default_segmentation_dataset(
107        raw_paths=image_paths,
108        raw_key=None,
109        label_paths=gt_paths,
110        label_key=None,
111        patch_shape=patch_shape,
112        is_seg_dataset=False,
113        **kwargs
114    )

Get the CVC-ColonDB dataset for polyp segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_cvc_colondb_loader( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], batch_size: int, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
117def get_cvc_colondb_loader(
118    path: Union[os.PathLike, str],
119    patch_shape: Tuple[int, int],
120    batch_size: int,
121    resize_inputs: bool = False,
122    download: bool = False,
123    **kwargs
124) -> DataLoader:
125    """Get the CVC-ColonDB dataloader for polyp segmentation.
126
127    Args:
128        path: Filepath to a folder where the data is downloaded for further processing.
129        patch_shape: The patch shape to use for training.
130        batch_size: The batch size for training.
131        resize_inputs: Whether to resize the inputs to the patch shape.
132        download: Whether to download the data if it is not present.
133        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
134
135    Returns:
136        The DataLoader.
137    """
138    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
139    dataset = get_cvc_colondb_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
140    return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)

Get the CVC-ColonDB dataloader for polyp segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • batch_size: The batch size for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.