torch_em.data.datasets.medical.cvc_colondb
The CVC-ColonDB dataset contains annotations for polyp segmentation in colonoscopy images.
The dataset consists of 380 still colonoscopy frames extracted from 15 different video sequences, each paired with a binary segmentation mask of the polyp region.
The dataset is located at https://www.kaggle.com/datasets/longvil/cvc-colondb. This is a mirror of the original CVC-ColonDB release from the Computer Vision Center (CVC), Barcelona, which is gated behind manual registration on the CVC-Colon website (https://pages.cvc.uab.es/CVC-Colon/).
This dataset is from the publication https://doi.org/10.1016/j.patcog.2012.03.002. Please cite it if you use this dataset for your research.
1"""The CVC-ColonDB dataset contains annotations for polyp segmentation in colonoscopy images. 2 3The dataset consists of 380 still colonoscopy frames extracted from 15 different video sequences, 4each paired with a binary segmentation mask of the polyp region. 5 6The dataset is located at https://www.kaggle.com/datasets/longvil/cvc-colondb. This is a mirror 7of the original CVC-ColonDB release from the Computer Vision Center (CVC), Barcelona, which is 8gated behind manual registration on the CVC-Colon website (https://pages.cvc.uab.es/CVC-Colon/). 9 10This dataset is from the publication https://doi.org/10.1016/j.patcog.2012.03.002. 11Please cite it if you use this dataset for your research. 12""" 13 14import os 15from glob import glob 16from natsort import natsorted 17from typing import Union, Tuple, List 18 19from torch.utils.data import Dataset, DataLoader 20 21import torch_em 22 23from .. import util 24 25 26KAGGLE_DATASET_NAME = "longvil/cvc-colondb" 27 28 29def get_cvc_colondb_data(path: Union[os.PathLike, str], download: bool = False) -> str: 30 """Download the CVC-ColonDB dataset. 31 32 Args: 33 path: Filepath to a folder where the data is downloaded for further processing. 34 download: Whether to download the data if it is not present. 35 36 Returns: 37 Filepath where the data is downloaded. 38 """ 39 data_dir = os.path.join(path, "CVC-ColonDB") 40 if os.path.exists(data_dir): 41 return data_dir 42 43 os.makedirs(path, exist_ok=True) 44 45 util.download_source_kaggle(path=path, dataset_name=KAGGLE_DATASET_NAME, download=download) 46 47 zip_path = os.path.join(path, "cvc-colondb.zip") 48 util.unzip(zip_path=zip_path, dst=path) 49 50 if not os.path.exists(data_dir): 51 raise RuntimeError(f"The dataset could not be found at '{path}' after extraction.") 52 53 return data_dir 54 55 56def get_cvc_colondb_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 57 """Get paths to the CVC-ColonDB data. 58 59 Args: 60 path: Filepath to a folder where the data is downloaded for further processing. 61 download: Whether to download the data if it is not present. 62 63 Returns: 64 List of filepaths for the image data. 65 List of filepaths for the label data. 66 """ 67 data_dir = get_cvc_colondb_data(path=path, download=download) 68 69 image_paths = natsorted(glob(os.path.join(data_dir, "images", "*.png"))) 70 gt_paths = natsorted(glob(os.path.join(data_dir, "masks", "*.png"))) 71 72 if len(image_paths) == 0 or len(image_paths) != len(gt_paths): 73 raise RuntimeError("Something went wrong with fetching the image and label paths.") 74 75 return image_paths, gt_paths 76 77 78def get_cvc_colondb_dataset( 79 path: Union[os.PathLike, str], 80 patch_shape: Tuple[int, int], 81 resize_inputs: bool = False, 82 download: bool = False, 83 **kwargs 84) -> Dataset: 85 """Get the CVC-ColonDB dataset for polyp segmentation. 86 87 Args: 88 path: Filepath to a folder where the data is downloaded for further processing. 89 patch_shape: The patch shape to use for training. 90 resize_inputs: Whether to resize the inputs to the patch shape. 91 download: Whether to download the data if it is not present. 92 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 93 94 Returns: 95 The segmentation dataset. 96 """ 97 image_paths, gt_paths = get_cvc_colondb_paths(path, download) 98 99 if resize_inputs: 100 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 101 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 102 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 103 ) 104 105 return torch_em.default_segmentation_dataset( 106 raw_paths=image_paths, 107 raw_key=None, 108 label_paths=gt_paths, 109 label_key=None, 110 patch_shape=patch_shape, 111 is_seg_dataset=False, 112 **kwargs 113 ) 114 115 116def get_cvc_colondb_loader( 117 path: Union[os.PathLike, str], 118 patch_shape: Tuple[int, int], 119 batch_size: int, 120 resize_inputs: bool = False, 121 download: bool = False, 122 **kwargs 123) -> DataLoader: 124 """Get the CVC-ColonDB dataloader for polyp segmentation. 125 126 Args: 127 path: Filepath to a folder where the data is downloaded for further processing. 128 patch_shape: The patch shape to use for training. 129 batch_size: The batch size for training. 130 resize_inputs: Whether to resize the inputs to the patch shape. 131 download: Whether to download the data if it is not present. 132 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 133 134 Returns: 135 The DataLoader. 136 """ 137 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 138 dataset = get_cvc_colondb_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 139 return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
30def get_cvc_colondb_data(path: Union[os.PathLike, str], download: bool = False) -> str: 31 """Download the CVC-ColonDB dataset. 32 33 Args: 34 path: Filepath to a folder where the data is downloaded for further processing. 35 download: Whether to download the data if it is not present. 36 37 Returns: 38 Filepath where the data is downloaded. 39 """ 40 data_dir = os.path.join(path, "CVC-ColonDB") 41 if os.path.exists(data_dir): 42 return data_dir 43 44 os.makedirs(path, exist_ok=True) 45 46 util.download_source_kaggle(path=path, dataset_name=KAGGLE_DATASET_NAME, download=download) 47 48 zip_path = os.path.join(path, "cvc-colondb.zip") 49 util.unzip(zip_path=zip_path, dst=path) 50 51 if not os.path.exists(data_dir): 52 raise RuntimeError(f"The dataset could not be found at '{path}' after extraction.") 53 54 return data_dir
Download the CVC-ColonDB dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
57def get_cvc_colondb_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 58 """Get paths to the CVC-ColonDB data. 59 60 Args: 61 path: Filepath to a folder where the data is downloaded for further processing. 62 download: Whether to download the data if it is not present. 63 64 Returns: 65 List of filepaths for the image data. 66 List of filepaths for the label data. 67 """ 68 data_dir = get_cvc_colondb_data(path=path, download=download) 69 70 image_paths = natsorted(glob(os.path.join(data_dir, "images", "*.png"))) 71 gt_paths = natsorted(glob(os.path.join(data_dir, "masks", "*.png"))) 72 73 if len(image_paths) == 0 or len(image_paths) != len(gt_paths): 74 raise RuntimeError("Something went wrong with fetching the image and label paths.") 75 76 return image_paths, gt_paths
Get paths to the CVC-ColonDB data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
79def get_cvc_colondb_dataset( 80 path: Union[os.PathLike, str], 81 patch_shape: Tuple[int, int], 82 resize_inputs: bool = False, 83 download: bool = False, 84 **kwargs 85) -> Dataset: 86 """Get the CVC-ColonDB dataset for polyp segmentation. 87 88 Args: 89 path: Filepath to a folder where the data is downloaded for further processing. 90 patch_shape: The patch shape to use for training. 91 resize_inputs: Whether to resize the inputs to the patch shape. 92 download: Whether to download the data if it is not present. 93 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 94 95 Returns: 96 The segmentation dataset. 97 """ 98 image_paths, gt_paths = get_cvc_colondb_paths(path, download) 99 100 if resize_inputs: 101 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 102 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 103 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 104 ) 105 106 return torch_em.default_segmentation_dataset( 107 raw_paths=image_paths, 108 raw_key=None, 109 label_paths=gt_paths, 110 label_key=None, 111 patch_shape=patch_shape, 112 is_seg_dataset=False, 113 **kwargs 114 )
Get the CVC-ColonDB dataset for polyp segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
117def get_cvc_colondb_loader( 118 path: Union[os.PathLike, str], 119 patch_shape: Tuple[int, int], 120 batch_size: int, 121 resize_inputs: bool = False, 122 download: bool = False, 123 **kwargs 124) -> DataLoader: 125 """Get the CVC-ColonDB dataloader for polyp segmentation. 126 127 Args: 128 path: Filepath to a folder where the data is downloaded for further processing. 129 patch_shape: The patch shape to use for training. 130 batch_size: The batch size for training. 131 resize_inputs: Whether to resize the inputs to the patch shape. 132 download: Whether to download the data if it is not present. 133 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 134 135 Returns: 136 The DataLoader. 137 """ 138 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 139 dataset = get_cvc_colondb_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 140 return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
Get the CVC-ColonDB dataloader for polyp segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- batch_size: The batch size for training.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.