torch_em.data.datasets.medical.bkai_igh_neopolyp

The BKAI-IGH NeoPolyp dataset contains annotations for semantic segmentation of neoplastic and non-neoplastic polyps in colonoscopy images.

NOTE: The ground-truth masks are stored as red (neoplastic polyp) and green (non-neoplastic polyp) regions on a black background, and the archive stores them as JPEG images. This means that lossy compression introduces off-palette colors along the region boundaries. We resolve this by assigning each pixel to whichever of the three reference colors (background, red, green) it is closest to, yielding semantic labels: 0 (background), 1 (non-neoplastic polyp) and 2 (neoplastic polyp).

NOTE: This dataset requires the Kaggle API. You need to install it via 'pip install kaggle' and set up an API token, see https://www.kaggle.com/docs/api. You also need to accept the competition rules on the Kaggle website (https://www.kaggle.com/c/bkai-igh-neopolyp/rules) before the download will succeed.

The dataset is located at https://www.kaggle.com/c/bkai-igh-neopolyp. This dataset is from the publication https://doi.org/10.1007/978-3-030-90436-4_2. Please cite it if you use this dataset for your research.

  1"""The BKAI-IGH NeoPolyp dataset contains annotations for semantic segmentation of neoplastic
  2and non-neoplastic polyps in colonoscopy images.
  3
  4NOTE: The ground-truth masks are stored as red (neoplastic polyp) and green (non-neoplastic
  5polyp) regions on a black background, and the archive stores them as JPEG images. This means
  6that lossy compression introduces off-palette colors along the region boundaries. We resolve
  7this by assigning each pixel to whichever of the three reference colors (background, red,
  8green) it is closest to, yielding semantic labels: 0 (background), 1 (non-neoplastic polyp)
  9and 2 (neoplastic polyp).
 10
 11NOTE: This dataset requires the Kaggle API. You need to install it via 'pip install kaggle'
 12and set up an API token, see https://www.kaggle.com/docs/api. You also need to accept the
 13competition rules on the Kaggle website (https://www.kaggle.com/c/bkai-igh-neopolyp/rules)
 14before the download will succeed.
 15
 16The dataset is located at https://www.kaggle.com/c/bkai-igh-neopolyp.
 17This dataset is from the publication https://doi.org/10.1007/978-3-030-90436-4_2.
 18Please cite it if you use this dataset for your research.
 19"""
 20
 21import os
 22from glob import glob
 23from tqdm import tqdm
 24from pathlib import Path
 25from natsort import natsorted
 26from typing import Union, Tuple, List
 27
 28import numpy as np
 29import imageio.v3 as imageio
 30
 31from torch.utils.data import Dataset, DataLoader
 32
 33import torch_em
 34
 35from .. import util
 36
 37
 38LABEL_COLORS = {
 39    0: (0, 0, 0),  # background
 40    1: (0, 255, 0),  # non-neoplastic polyp
 41    2: (255, 0, 0),  # neoplastic polyp
 42}
 43
 44
 45def get_bkai_igh_neopolyp_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 46    """Download the BKAI-IGH NeoPolyp dataset.
 47
 48    Args:
 49        path: Filepath to a folder where the data is downloaded for further processing.
 50        download: Whether to download the data if it is not present.
 51
 52    Returns:
 53        Filepath where the data is downloaded.
 54    """
 55    data_dir = os.path.join(path, "train")
 56    if os.path.exists(data_dir):
 57        return path
 58
 59    os.makedirs(path, exist_ok=True)
 60
 61    zip_path = os.path.join(path, "bkai-igh-neopolyp.zip")
 62    util.download_source_kaggle(path=path, dataset_name="bkai-igh-neopolyp", download=download, competition=True)
 63    util.unzip(zip_path=zip_path, dst=path)
 64
 65    # The competition bundle ships 'train', 'train_gt' and 'test' as nested zip archives.
 66    for name in ["train", "train_gt", "test"]:
 67        nested_zip = os.path.join(path, f"{name}.zip")
 68        if os.path.exists(nested_zip):
 69            util.unzip(zip_path=nested_zip, dst=path)
 70
 71    return path
 72
 73
 74def get_bkai_igh_neopolyp_paths(
 75    path: Union[os.PathLike, str], download: bool = False
 76) -> Tuple[List[str], List[str]]:
 77    """Get paths to the BKAI-IGH NeoPolyp data.
 78
 79    Args:
 80        path: Filepath to a folder where the data is downloaded for further processing.
 81        download: Whether to download the data if it is not present.
 82
 83    Returns:
 84        List of filepaths for the image data.
 85        List of filepaths for the label data.
 86    """
 87    data_dir = get_bkai_igh_neopolyp_data(path=path, download=download)
 88
 89    image_paths = natsorted(glob(os.path.join(data_dir, "train", "train", "*.jpeg")))
 90    gt_paths = natsorted(glob(os.path.join(data_dir, "train_gt", "train_gt", "*.jpeg")))
 91
 92    neu_gt_dir = os.path.join(data_dir, "train_gt", "preprocessed")
 93    os.makedirs(neu_gt_dir, exist_ok=True)
 94
 95    reference_colors = np.array(list(LABEL_COLORS.values()))
 96
 97    neu_gt_paths = []
 98    for gt_path in tqdm(gt_paths, desc="Preprocessing labels"):
 99        neu_gt_path = os.path.join(neu_gt_dir, f"{Path(gt_path).stem}.tif")
100        neu_gt_paths.append(neu_gt_path)
101        if os.path.exists(neu_gt_path):
102            continue
103
104        gt = imageio.imread(gt_path)[..., :3].astype("float32")
105        distances = np.linalg.norm(gt[..., None, :] - reference_colors[None, None, :, :], axis=-1)
106        semantic_gt = np.argmin(distances, axis=-1).astype("uint8")
107        imageio.imwrite(neu_gt_path, semantic_gt, compression="zlib")
108
109    return image_paths, neu_gt_paths
110
111
112def get_bkai_igh_neopolyp_dataset(
113    path: Union[os.PathLike, str],
114    patch_shape: Tuple[int, int],
115    resize_inputs: bool = False,
116    download: bool = False,
117    **kwargs
118) -> Dataset:
119    """Get the BKAI-IGH NeoPolyp dataset for polyp segmentation.
120
121    Args:
122        path: Filepath to a folder where the data is downloaded for further processing.
123        patch_shape: The patch shape to use for training.
124        resize_inputs: Whether to resize the inputs to the patch shape.
125        download: Whether to download the data if it is not present.
126        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
127
128    Returns:
129        The segmentation dataset.
130    """
131    image_paths, gt_paths = get_bkai_igh_neopolyp_paths(path, download)
132
133    if resize_inputs:
134        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
135        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
136            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
137        )
138
139    return torch_em.default_segmentation_dataset(
140        raw_paths=image_paths,
141        raw_key=None,
142        label_paths=gt_paths,
143        label_key=None,
144        patch_shape=patch_shape,
145        is_seg_dataset=False,
146        **kwargs
147    )
148
149
150def get_bkai_igh_neopolyp_loader(
151    path: Union[os.PathLike, str],
152    patch_shape: Tuple[int, int],
153    batch_size: int,
154    resize_inputs: bool = False,
155    download: bool = False,
156    **kwargs
157) -> DataLoader:
158    """Get the BKAI-IGH NeoPolyp dataloader for polyp segmentation.
159
160    Args:
161        path: Filepath to a folder where the data is downloaded for further processing.
162        patch_shape: The patch shape to use for training.
163        batch_size: The batch size for training.
164        resize_inputs: Whether to resize the inputs to the patch shape.
165        download: Whether to download the data if it is not present.
166        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
167
168    Returns:
169        The DataLoader.
170    """
171    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
172    dataset = get_bkai_igh_neopolyp_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
173    return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
LABEL_COLORS = {0: (0, 0, 0), 1: (0, 255, 0), 2: (255, 0, 0)}
def get_bkai_igh_neopolyp_data(path: Union[os.PathLike, str], download: bool = False) -> str:
46def get_bkai_igh_neopolyp_data(path: Union[os.PathLike, str], download: bool = False) -> str:
47    """Download the BKAI-IGH NeoPolyp dataset.
48
49    Args:
50        path: Filepath to a folder where the data is downloaded for further processing.
51        download: Whether to download the data if it is not present.
52
53    Returns:
54        Filepath where the data is downloaded.
55    """
56    data_dir = os.path.join(path, "train")
57    if os.path.exists(data_dir):
58        return path
59
60    os.makedirs(path, exist_ok=True)
61
62    zip_path = os.path.join(path, "bkai-igh-neopolyp.zip")
63    util.download_source_kaggle(path=path, dataset_name="bkai-igh-neopolyp", download=download, competition=True)
64    util.unzip(zip_path=zip_path, dst=path)
65
66    # The competition bundle ships 'train', 'train_gt' and 'test' as nested zip archives.
67    for name in ["train", "train_gt", "test"]:
68        nested_zip = os.path.join(path, f"{name}.zip")
69        if os.path.exists(nested_zip):
70            util.unzip(zip_path=nested_zip, dst=path)
71
72    return path

Download the BKAI-IGH NeoPolyp dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_bkai_igh_neopolyp_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 75def get_bkai_igh_neopolyp_paths(
 76    path: Union[os.PathLike, str], download: bool = False
 77) -> Tuple[List[str], List[str]]:
 78    """Get paths to the BKAI-IGH NeoPolyp data.
 79
 80    Args:
 81        path: Filepath to a folder where the data is downloaded for further processing.
 82        download: Whether to download the data if it is not present.
 83
 84    Returns:
 85        List of filepaths for the image data.
 86        List of filepaths for the label data.
 87    """
 88    data_dir = get_bkai_igh_neopolyp_data(path=path, download=download)
 89
 90    image_paths = natsorted(glob(os.path.join(data_dir, "train", "train", "*.jpeg")))
 91    gt_paths = natsorted(glob(os.path.join(data_dir, "train_gt", "train_gt", "*.jpeg")))
 92
 93    neu_gt_dir = os.path.join(data_dir, "train_gt", "preprocessed")
 94    os.makedirs(neu_gt_dir, exist_ok=True)
 95
 96    reference_colors = np.array(list(LABEL_COLORS.values()))
 97
 98    neu_gt_paths = []
 99    for gt_path in tqdm(gt_paths, desc="Preprocessing labels"):
100        neu_gt_path = os.path.join(neu_gt_dir, f"{Path(gt_path).stem}.tif")
101        neu_gt_paths.append(neu_gt_path)
102        if os.path.exists(neu_gt_path):
103            continue
104
105        gt = imageio.imread(gt_path)[..., :3].astype("float32")
106        distances = np.linalg.norm(gt[..., None, :] - reference_colors[None, None, :, :], axis=-1)
107        semantic_gt = np.argmin(distances, axis=-1).astype("uint8")
108        imageio.imwrite(neu_gt_path, semantic_gt, compression="zlib")
109
110    return image_paths, neu_gt_paths

Get paths to the BKAI-IGH NeoPolyp data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_bkai_igh_neopolyp_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
113def get_bkai_igh_neopolyp_dataset(
114    path: Union[os.PathLike, str],
115    patch_shape: Tuple[int, int],
116    resize_inputs: bool = False,
117    download: bool = False,
118    **kwargs
119) -> Dataset:
120    """Get the BKAI-IGH NeoPolyp dataset for polyp segmentation.
121
122    Args:
123        path: Filepath to a folder where the data is downloaded for further processing.
124        patch_shape: The patch shape to use for training.
125        resize_inputs: Whether to resize the inputs to the patch shape.
126        download: Whether to download the data if it is not present.
127        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
128
129    Returns:
130        The segmentation dataset.
131    """
132    image_paths, gt_paths = get_bkai_igh_neopolyp_paths(path, download)
133
134    if resize_inputs:
135        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
136        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
137            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
138        )
139
140    return torch_em.default_segmentation_dataset(
141        raw_paths=image_paths,
142        raw_key=None,
143        label_paths=gt_paths,
144        label_key=None,
145        patch_shape=patch_shape,
146        is_seg_dataset=False,
147        **kwargs
148    )

Get the BKAI-IGH NeoPolyp dataset for polyp segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_bkai_igh_neopolyp_loader( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], batch_size: int, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
151def get_bkai_igh_neopolyp_loader(
152    path: Union[os.PathLike, str],
153    patch_shape: Tuple[int, int],
154    batch_size: int,
155    resize_inputs: bool = False,
156    download: bool = False,
157    **kwargs
158) -> DataLoader:
159    """Get the BKAI-IGH NeoPolyp dataloader for polyp segmentation.
160
161    Args:
162        path: Filepath to a folder where the data is downloaded for further processing.
163        patch_shape: The patch shape to use for training.
164        batch_size: The batch size for training.
165        resize_inputs: Whether to resize the inputs to the patch shape.
166        download: Whether to download the data if it is not present.
167        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
168
169    Returns:
170        The DataLoader.
171    """
172    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
173    dataset = get_bkai_igh_neopolyp_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
174    return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)

Get the BKAI-IGH NeoPolyp dataloader for polyp segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • batch_size: The batch size for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.