torch_em.data.datasets.medical.bkai_igh_neopolyp
The BKAI-IGH NeoPolyp dataset contains annotations for semantic segmentation of neoplastic and non-neoplastic polyps in colonoscopy images.
NOTE: The ground-truth masks are stored as red (neoplastic polyp) and green (non-neoplastic polyp) regions on a black background, and the archive stores them as JPEG images. This means that lossy compression introduces off-palette colors along the region boundaries. We resolve this by assigning each pixel to whichever of the three reference colors (background, red, green) it is closest to, yielding semantic labels: 0 (background), 1 (non-neoplastic polyp) and 2 (neoplastic polyp).
NOTE: This dataset requires the Kaggle API. You need to install it via 'pip install kaggle' and set up an API token, see https://www.kaggle.com/docs/api. You also need to accept the competition rules on the Kaggle website (https://www.kaggle.com/c/bkai-igh-neopolyp/rules) before the download will succeed.
The dataset is located at https://www.kaggle.com/c/bkai-igh-neopolyp. This dataset is from the publication https://doi.org/10.1007/978-3-030-90436-4_2. Please cite it if you use this dataset for your research.
1"""The BKAI-IGH NeoPolyp dataset contains annotations for semantic segmentation of neoplastic 2and non-neoplastic polyps in colonoscopy images. 3 4NOTE: The ground-truth masks are stored as red (neoplastic polyp) and green (non-neoplastic 5polyp) regions on a black background, and the archive stores them as JPEG images. This means 6that lossy compression introduces off-palette colors along the region boundaries. We resolve 7this by assigning each pixel to whichever of the three reference colors (background, red, 8green) it is closest to, yielding semantic labels: 0 (background), 1 (non-neoplastic polyp) 9and 2 (neoplastic polyp). 10 11NOTE: This dataset requires the Kaggle API. You need to install it via 'pip install kaggle' 12and set up an API token, see https://www.kaggle.com/docs/api. You also need to accept the 13competition rules on the Kaggle website (https://www.kaggle.com/c/bkai-igh-neopolyp/rules) 14before the download will succeed. 15 16The dataset is located at https://www.kaggle.com/c/bkai-igh-neopolyp. 17This dataset is from the publication https://doi.org/10.1007/978-3-030-90436-4_2. 18Please cite it if you use this dataset for your research. 19""" 20 21import os 22from glob import glob 23from tqdm import tqdm 24from pathlib import Path 25from natsort import natsorted 26from typing import Union, Tuple, List 27 28import numpy as np 29import imageio.v3 as imageio 30 31from torch.utils.data import Dataset, DataLoader 32 33import torch_em 34 35from .. import util 36 37 38LABEL_COLORS = { 39 0: (0, 0, 0), # background 40 1: (0, 255, 0), # non-neoplastic polyp 41 2: (255, 0, 0), # neoplastic polyp 42} 43 44 45def get_bkai_igh_neopolyp_data(path: Union[os.PathLike, str], download: bool = False) -> str: 46 """Download the BKAI-IGH NeoPolyp dataset. 47 48 Args: 49 path: Filepath to a folder where the data is downloaded for further processing. 50 download: Whether to download the data if it is not present. 51 52 Returns: 53 Filepath where the data is downloaded. 54 """ 55 data_dir = os.path.join(path, "train") 56 if os.path.exists(data_dir): 57 return path 58 59 os.makedirs(path, exist_ok=True) 60 61 zip_path = os.path.join(path, "bkai-igh-neopolyp.zip") 62 util.download_source_kaggle(path=path, dataset_name="bkai-igh-neopolyp", download=download, competition=True) 63 util.unzip(zip_path=zip_path, dst=path) 64 65 # The competition bundle ships 'train', 'train_gt' and 'test' as nested zip archives. 66 for name in ["train", "train_gt", "test"]: 67 nested_zip = os.path.join(path, f"{name}.zip") 68 if os.path.exists(nested_zip): 69 util.unzip(zip_path=nested_zip, dst=path) 70 71 return path 72 73 74def get_bkai_igh_neopolyp_paths( 75 path: Union[os.PathLike, str], download: bool = False 76) -> Tuple[List[str], List[str]]: 77 """Get paths to the BKAI-IGH NeoPolyp data. 78 79 Args: 80 path: Filepath to a folder where the data is downloaded for further processing. 81 download: Whether to download the data if it is not present. 82 83 Returns: 84 List of filepaths for the image data. 85 List of filepaths for the label data. 86 """ 87 data_dir = get_bkai_igh_neopolyp_data(path=path, download=download) 88 89 image_paths = natsorted(glob(os.path.join(data_dir, "train", "train", "*.jpeg"))) 90 gt_paths = natsorted(glob(os.path.join(data_dir, "train_gt", "train_gt", "*.jpeg"))) 91 92 neu_gt_dir = os.path.join(data_dir, "train_gt", "preprocessed") 93 os.makedirs(neu_gt_dir, exist_ok=True) 94 95 reference_colors = np.array(list(LABEL_COLORS.values())) 96 97 neu_gt_paths = [] 98 for gt_path in tqdm(gt_paths, desc="Preprocessing labels"): 99 neu_gt_path = os.path.join(neu_gt_dir, f"{Path(gt_path).stem}.tif") 100 neu_gt_paths.append(neu_gt_path) 101 if os.path.exists(neu_gt_path): 102 continue 103 104 gt = imageio.imread(gt_path)[..., :3].astype("float32") 105 distances = np.linalg.norm(gt[..., None, :] - reference_colors[None, None, :, :], axis=-1) 106 semantic_gt = np.argmin(distances, axis=-1).astype("uint8") 107 imageio.imwrite(neu_gt_path, semantic_gt, compression="zlib") 108 109 return image_paths, neu_gt_paths 110 111 112def get_bkai_igh_neopolyp_dataset( 113 path: Union[os.PathLike, str], 114 patch_shape: Tuple[int, int], 115 resize_inputs: bool = False, 116 download: bool = False, 117 **kwargs 118) -> Dataset: 119 """Get the BKAI-IGH NeoPolyp dataset for polyp segmentation. 120 121 Args: 122 path: Filepath to a folder where the data is downloaded for further processing. 123 patch_shape: The patch shape to use for training. 124 resize_inputs: Whether to resize the inputs to the patch shape. 125 download: Whether to download the data if it is not present. 126 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 127 128 Returns: 129 The segmentation dataset. 130 """ 131 image_paths, gt_paths = get_bkai_igh_neopolyp_paths(path, download) 132 133 if resize_inputs: 134 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 135 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 136 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 137 ) 138 139 return torch_em.default_segmentation_dataset( 140 raw_paths=image_paths, 141 raw_key=None, 142 label_paths=gt_paths, 143 label_key=None, 144 patch_shape=patch_shape, 145 is_seg_dataset=False, 146 **kwargs 147 ) 148 149 150def get_bkai_igh_neopolyp_loader( 151 path: Union[os.PathLike, str], 152 patch_shape: Tuple[int, int], 153 batch_size: int, 154 resize_inputs: bool = False, 155 download: bool = False, 156 **kwargs 157) -> DataLoader: 158 """Get the BKAI-IGH NeoPolyp dataloader for polyp segmentation. 159 160 Args: 161 path: Filepath to a folder where the data is downloaded for further processing. 162 patch_shape: The patch shape to use for training. 163 batch_size: The batch size for training. 164 resize_inputs: Whether to resize the inputs to the patch shape. 165 download: Whether to download the data if it is not present. 166 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 167 168 Returns: 169 The DataLoader. 170 """ 171 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 172 dataset = get_bkai_igh_neopolyp_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 173 return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
46def get_bkai_igh_neopolyp_data(path: Union[os.PathLike, str], download: bool = False) -> str: 47 """Download the BKAI-IGH NeoPolyp dataset. 48 49 Args: 50 path: Filepath to a folder where the data is downloaded for further processing. 51 download: Whether to download the data if it is not present. 52 53 Returns: 54 Filepath where the data is downloaded. 55 """ 56 data_dir = os.path.join(path, "train") 57 if os.path.exists(data_dir): 58 return path 59 60 os.makedirs(path, exist_ok=True) 61 62 zip_path = os.path.join(path, "bkai-igh-neopolyp.zip") 63 util.download_source_kaggle(path=path, dataset_name="bkai-igh-neopolyp", download=download, competition=True) 64 util.unzip(zip_path=zip_path, dst=path) 65 66 # The competition bundle ships 'train', 'train_gt' and 'test' as nested zip archives. 67 for name in ["train", "train_gt", "test"]: 68 nested_zip = os.path.join(path, f"{name}.zip") 69 if os.path.exists(nested_zip): 70 util.unzip(zip_path=nested_zip, dst=path) 71 72 return path
Download the BKAI-IGH NeoPolyp dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
75def get_bkai_igh_neopolyp_paths( 76 path: Union[os.PathLike, str], download: bool = False 77) -> Tuple[List[str], List[str]]: 78 """Get paths to the BKAI-IGH NeoPolyp data. 79 80 Args: 81 path: Filepath to a folder where the data is downloaded for further processing. 82 download: Whether to download the data if it is not present. 83 84 Returns: 85 List of filepaths for the image data. 86 List of filepaths for the label data. 87 """ 88 data_dir = get_bkai_igh_neopolyp_data(path=path, download=download) 89 90 image_paths = natsorted(glob(os.path.join(data_dir, "train", "train", "*.jpeg"))) 91 gt_paths = natsorted(glob(os.path.join(data_dir, "train_gt", "train_gt", "*.jpeg"))) 92 93 neu_gt_dir = os.path.join(data_dir, "train_gt", "preprocessed") 94 os.makedirs(neu_gt_dir, exist_ok=True) 95 96 reference_colors = np.array(list(LABEL_COLORS.values())) 97 98 neu_gt_paths = [] 99 for gt_path in tqdm(gt_paths, desc="Preprocessing labels"): 100 neu_gt_path = os.path.join(neu_gt_dir, f"{Path(gt_path).stem}.tif") 101 neu_gt_paths.append(neu_gt_path) 102 if os.path.exists(neu_gt_path): 103 continue 104 105 gt = imageio.imread(gt_path)[..., :3].astype("float32") 106 distances = np.linalg.norm(gt[..., None, :] - reference_colors[None, None, :, :], axis=-1) 107 semantic_gt = np.argmin(distances, axis=-1).astype("uint8") 108 imageio.imwrite(neu_gt_path, semantic_gt, compression="zlib") 109 110 return image_paths, neu_gt_paths
Get paths to the BKAI-IGH NeoPolyp data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
113def get_bkai_igh_neopolyp_dataset( 114 path: Union[os.PathLike, str], 115 patch_shape: Tuple[int, int], 116 resize_inputs: bool = False, 117 download: bool = False, 118 **kwargs 119) -> Dataset: 120 """Get the BKAI-IGH NeoPolyp dataset for polyp segmentation. 121 122 Args: 123 path: Filepath to a folder where the data is downloaded for further processing. 124 patch_shape: The patch shape to use for training. 125 resize_inputs: Whether to resize the inputs to the patch shape. 126 download: Whether to download the data if it is not present. 127 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 128 129 Returns: 130 The segmentation dataset. 131 """ 132 image_paths, gt_paths = get_bkai_igh_neopolyp_paths(path, download) 133 134 if resize_inputs: 135 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 136 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 137 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 138 ) 139 140 return torch_em.default_segmentation_dataset( 141 raw_paths=image_paths, 142 raw_key=None, 143 label_paths=gt_paths, 144 label_key=None, 145 patch_shape=patch_shape, 146 is_seg_dataset=False, 147 **kwargs 148 )
Get the BKAI-IGH NeoPolyp dataset for polyp segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
151def get_bkai_igh_neopolyp_loader( 152 path: Union[os.PathLike, str], 153 patch_shape: Tuple[int, int], 154 batch_size: int, 155 resize_inputs: bool = False, 156 download: bool = False, 157 **kwargs 158) -> DataLoader: 159 """Get the BKAI-IGH NeoPolyp dataloader for polyp segmentation. 160 161 Args: 162 path: Filepath to a folder where the data is downloaded for further processing. 163 patch_shape: The patch shape to use for training. 164 batch_size: The batch size for training. 165 resize_inputs: Whether to resize the inputs to the patch shape. 166 download: Whether to download the data if it is not present. 167 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 168 169 Returns: 170 The DataLoader. 171 """ 172 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 173 dataset = get_bkai_igh_neopolyp_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 174 return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
Get the BKAI-IGH NeoPolyp dataloader for polyp segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- batch_size: The batch size for training.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.