torch_em.data.datasets.medical.cvc_endoscenestill

The CVC-EndoSceneStill dataset contains annotations for polyp segmentation in colonoscopy images.

NOTE: The full CVC-EndoSceneStill release (912 stills split into train / validation / test, with additional semantic classes for specular highlights and the lumen) is gated behind manual registration on the CVC-Colon website (https://pages.cvc.uab.es/CVC-Colon/index.php/databases/cvc-endoscenestill/). We instead provide the openly mirrored "CVC-300" subset, which corresponds to the 60-image test split of CVC-EndoSceneStill and only ships binary polyp masks. This subset is the one commonly used as a polyp segmentation benchmark (e.g. in the PraNet line of work).

The dataset is located at https://www.kaggle.com/datasets/nourabentaher/cvc-300.

This dataset is from the publication https://doi.org/10.1155/2017/4037190. Please cite it if you use this dataset for your research.

  1"""The CVC-EndoSceneStill dataset contains annotations for polyp segmentation in colonoscopy images.
  2
  3NOTE: The full CVC-EndoSceneStill release (912 stills split into train / validation / test, with
  4additional semantic classes for specular highlights and the lumen) is gated behind manual
  5registration on the CVC-Colon website (https://pages.cvc.uab.es/CVC-Colon/index.php/databases/cvc-endoscenestill/).
  6We instead provide the openly mirrored "CVC-300" subset, which corresponds to the 60-image test
  7split of CVC-EndoSceneStill and only ships binary polyp masks. This subset is the one commonly
  8used as a polyp segmentation benchmark (e.g. in the PraNet line of work).
  9
 10The dataset is located at https://www.kaggle.com/datasets/nourabentaher/cvc-300.
 11
 12This dataset is from the publication https://doi.org/10.1155/2017/4037190.
 13Please cite it if you use this dataset for your research.
 14"""
 15
 16import os
 17from glob import glob
 18from tqdm import tqdm
 19from pathlib import Path
 20from natsort import natsorted
 21from typing import Union, Tuple, List
 22
 23import numpy as np
 24import imageio.v3 as imageio
 25
 26from torch.utils.data import Dataset, DataLoader
 27
 28import torch_em
 29
 30from .. import util
 31
 32
 33KAGGLE_DATASET_NAME = "nourabentaher/cvc-300"
 34
 35
 36def get_cvc_endoscenestill_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 37    """Download the CVC-EndoSceneStill (CVC-300 subset) dataset.
 38
 39    Args:
 40        path: Filepath to a folder where the data is downloaded for further processing.
 41        download: Whether to download the data if it is not present.
 42
 43    Returns:
 44        Filepath where the data is downloaded.
 45    """
 46    data_dir = os.path.join(path, "CVC-300")
 47    if os.path.exists(data_dir):
 48        return data_dir
 49
 50    os.makedirs(path, exist_ok=True)
 51
 52    util.download_source_kaggle(path=path, dataset_name=KAGGLE_DATASET_NAME, download=download)
 53
 54    zip_path = os.path.join(path, "cvc-300.zip")
 55    util.unzip(zip_path=zip_path, dst=path)
 56
 57    if not os.path.exists(data_dir):
 58        raise RuntimeError(f"The dataset could not be found at '{path}' after extraction.")
 59
 60    return data_dir
 61
 62
 63def get_cvc_endoscenestill_paths(
 64    path: Union[os.PathLike, str], download: bool = False
 65) -> Tuple[List[str], List[str]]:
 66    """Get paths to the CVC-EndoSceneStill (CVC-300 subset) data.
 67
 68    Args:
 69        path: Filepath to a folder where the data is downloaded for further processing.
 70        download: Whether to download the data if it is not present.
 71
 72    Returns:
 73        List of filepaths for the image data.
 74        List of filepaths for the label data.
 75    """
 76    data_dir = get_cvc_endoscenestill_data(path=path, download=download)
 77
 78    image_paths = natsorted(glob(os.path.join(data_dir, "images", "*.png")))
 79    mask_paths = natsorted(glob(os.path.join(data_dir, "masks", "*.png")))
 80
 81    if len(image_paths) == 0 or len(image_paths) != len(mask_paths):
 82        raise RuntimeError("Something went wrong with fetching the image and label paths.")
 83
 84    neu_gt_dir = os.path.join(data_dir, "masks", "preprocessed")
 85    os.makedirs(neu_gt_dir, exist_ok=True)
 86
 87    gt_paths = []
 88    for mask_path in tqdm(mask_paths, desc="Preprocessing labels"):
 89        gt_path = os.path.join(neu_gt_dir, f"{Path(mask_path).stem}.tif")
 90        gt_paths.append(gt_path)
 91        if os.path.exists(gt_path):
 92            continue
 93
 94        mask = imageio.imread(mask_path)
 95        if mask.ndim == 3:
 96            mask = np.mean(mask, axis=-1)
 97        mask = (mask >= 128).astype("uint8")
 98        imageio.imwrite(gt_path, mask, compression="zlib")
 99
100    return image_paths, gt_paths
101
102
103def get_cvc_endoscenestill_dataset(
104    path: Union[os.PathLike, str],
105    patch_shape: Tuple[int, int],
106    resize_inputs: bool = False,
107    download: bool = False,
108    **kwargs
109) -> Dataset:
110    """Get the CVC-EndoSceneStill (CVC-300 subset) dataset for polyp segmentation.
111
112    Args:
113        path: Filepath to a folder where the data is downloaded for further processing.
114        patch_shape: The patch shape to use for training.
115        resize_inputs: Whether to resize the inputs to the patch shape.
116        download: Whether to download the data if it is not present.
117        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
118
119    Returns:
120        The segmentation dataset.
121    """
122    image_paths, gt_paths = get_cvc_endoscenestill_paths(path, download)
123
124    if resize_inputs:
125        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
126        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
127            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
128        )
129
130    return torch_em.default_segmentation_dataset(
131        raw_paths=image_paths,
132        raw_key=None,
133        label_paths=gt_paths,
134        label_key=None,
135        patch_shape=patch_shape,
136        is_seg_dataset=False,
137        **kwargs
138    )
139
140
141def get_cvc_endoscenestill_loader(
142    path: Union[os.PathLike, str],
143    patch_shape: Tuple[int, int],
144    batch_size: int,
145    resize_inputs: bool = False,
146    download: bool = False,
147    **kwargs
148) -> DataLoader:
149    """Get the CVC-EndoSceneStill (CVC-300 subset) dataloader for polyp segmentation.
150
151    Args:
152        path: Filepath to a folder where the data is downloaded for further processing.
153        patch_shape: The patch shape to use for training.
154        batch_size: The batch size for training.
155        resize_inputs: Whether to resize the inputs to the patch shape.
156        download: Whether to download the data if it is not present.
157        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
158
159    Returns:
160        The DataLoader.
161    """
162    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
163    dataset = get_cvc_endoscenestill_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
164    return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
KAGGLE_DATASET_NAME = 'nourabentaher/cvc-300'
def get_cvc_endoscenestill_data(path: Union[os.PathLike, str], download: bool = False) -> str:
37def get_cvc_endoscenestill_data(path: Union[os.PathLike, str], download: bool = False) -> str:
38    """Download the CVC-EndoSceneStill (CVC-300 subset) dataset.
39
40    Args:
41        path: Filepath to a folder where the data is downloaded for further processing.
42        download: Whether to download the data if it is not present.
43
44    Returns:
45        Filepath where the data is downloaded.
46    """
47    data_dir = os.path.join(path, "CVC-300")
48    if os.path.exists(data_dir):
49        return data_dir
50
51    os.makedirs(path, exist_ok=True)
52
53    util.download_source_kaggle(path=path, dataset_name=KAGGLE_DATASET_NAME, download=download)
54
55    zip_path = os.path.join(path, "cvc-300.zip")
56    util.unzip(zip_path=zip_path, dst=path)
57
58    if not os.path.exists(data_dir):
59        raise RuntimeError(f"The dataset could not be found at '{path}' after extraction.")
60
61    return data_dir

Download the CVC-EndoSceneStill (CVC-300 subset) dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_cvc_endoscenestill_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 64def get_cvc_endoscenestill_paths(
 65    path: Union[os.PathLike, str], download: bool = False
 66) -> Tuple[List[str], List[str]]:
 67    """Get paths to the CVC-EndoSceneStill (CVC-300 subset) data.
 68
 69    Args:
 70        path: Filepath to a folder where the data is downloaded for further processing.
 71        download: Whether to download the data if it is not present.
 72
 73    Returns:
 74        List of filepaths for the image data.
 75        List of filepaths for the label data.
 76    """
 77    data_dir = get_cvc_endoscenestill_data(path=path, download=download)
 78
 79    image_paths = natsorted(glob(os.path.join(data_dir, "images", "*.png")))
 80    mask_paths = natsorted(glob(os.path.join(data_dir, "masks", "*.png")))
 81
 82    if len(image_paths) == 0 or len(image_paths) != len(mask_paths):
 83        raise RuntimeError("Something went wrong with fetching the image and label paths.")
 84
 85    neu_gt_dir = os.path.join(data_dir, "masks", "preprocessed")
 86    os.makedirs(neu_gt_dir, exist_ok=True)
 87
 88    gt_paths = []
 89    for mask_path in tqdm(mask_paths, desc="Preprocessing labels"):
 90        gt_path = os.path.join(neu_gt_dir, f"{Path(mask_path).stem}.tif")
 91        gt_paths.append(gt_path)
 92        if os.path.exists(gt_path):
 93            continue
 94
 95        mask = imageio.imread(mask_path)
 96        if mask.ndim == 3:
 97            mask = np.mean(mask, axis=-1)
 98        mask = (mask >= 128).astype("uint8")
 99        imageio.imwrite(gt_path, mask, compression="zlib")
100
101    return image_paths, gt_paths

Get paths to the CVC-EndoSceneStill (CVC-300 subset) data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_cvc_endoscenestill_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
104def get_cvc_endoscenestill_dataset(
105    path: Union[os.PathLike, str],
106    patch_shape: Tuple[int, int],
107    resize_inputs: bool = False,
108    download: bool = False,
109    **kwargs
110) -> Dataset:
111    """Get the CVC-EndoSceneStill (CVC-300 subset) dataset for polyp segmentation.
112
113    Args:
114        path: Filepath to a folder where the data is downloaded for further processing.
115        patch_shape: The patch shape to use for training.
116        resize_inputs: Whether to resize the inputs to the patch shape.
117        download: Whether to download the data if it is not present.
118        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
119
120    Returns:
121        The segmentation dataset.
122    """
123    image_paths, gt_paths = get_cvc_endoscenestill_paths(path, download)
124
125    if resize_inputs:
126        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
127        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
128            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
129        )
130
131    return torch_em.default_segmentation_dataset(
132        raw_paths=image_paths,
133        raw_key=None,
134        label_paths=gt_paths,
135        label_key=None,
136        patch_shape=patch_shape,
137        is_seg_dataset=False,
138        **kwargs
139    )

Get the CVC-EndoSceneStill (CVC-300 subset) dataset for polyp segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_cvc_endoscenestill_loader( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], batch_size: int, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
142def get_cvc_endoscenestill_loader(
143    path: Union[os.PathLike, str],
144    patch_shape: Tuple[int, int],
145    batch_size: int,
146    resize_inputs: bool = False,
147    download: bool = False,
148    **kwargs
149) -> DataLoader:
150    """Get the CVC-EndoSceneStill (CVC-300 subset) dataloader for polyp segmentation.
151
152    Args:
153        path: Filepath to a folder where the data is downloaded for further processing.
154        patch_shape: The patch shape to use for training.
155        batch_size: The batch size for training.
156        resize_inputs: Whether to resize the inputs to the patch shape.
157        download: Whether to download the data if it is not present.
158        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
159
160    Returns:
161        The DataLoader.
162    """
163    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
164    dataset = get_cvc_endoscenestill_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
165    return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)

Get the CVC-EndoSceneStill (CVC-300 subset) dataloader for polyp segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • batch_size: The batch size for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.