torch_em.data.datasets.medical.edd2020

The EDD2020 dataset contains annotations for multi-class disease segmentation in gastrointestinal endoscopy images.

The dataset consists of 386 endoscopy frames from 5 different institutions and multiple GI organs (esophagus, stomach, colon), collected with white light, narrow-band imaging and chromoendoscopy. Each image is annotated with pixel-level masks for up to 5 disease classes: non-dysplastic Barrett's esophagus (BE), suspicious lesions, high-grade dysplasia (HGD), cancer, and polyp. The official challenge data is gated behind manual registration on https://edd2020.grand-challenge.org, so we instead use the openly mirrored copy at https://www.kaggle.com/datasets/orvile/edd2020-endoscopy-detection-and-segmentation, which matches the official release (386 images, same organizers and class structure). NOTE: the original data release states the license as CC BY-NC-SA 4.0, while the Kaggle mirror lists it as CC BY 4.0; please check the current license terms before using this data.

This dataset is from the publication https://doi.org/10.1016/j.media.2021.102002. Please cite it if you use this dataset for your research.

  1"""The EDD2020 dataset contains annotations for multi-class disease segmentation in
  2gastrointestinal endoscopy images.
  3
  4The dataset consists of 386 endoscopy frames from 5 different institutions and multiple
  5GI organs (esophagus, stomach, colon), collected with white light, narrow-band imaging and
  6chromoendoscopy. Each image is annotated with pixel-level masks for up to 5 disease classes:
  7non-dysplastic Barrett's esophagus (BE), suspicious lesions, high-grade dysplasia (HGD),
  8cancer, and polyp. The official challenge data is gated behind manual registration on
  9https://edd2020.grand-challenge.org, so we instead use the openly mirrored copy at
 10https://www.kaggle.com/datasets/orvile/edd2020-endoscopy-detection-and-segmentation, which
 11matches the official release (386 images, same organizers and class structure). NOTE: the
 12original data release states the license as CC BY-NC-SA 4.0, while the Kaggle mirror lists it
 13as CC BY 4.0; please check the current license terms before using this data.
 14
 15This dataset is from the publication https://doi.org/10.1016/j.media.2021.102002.
 16Please cite it if you use this dataset for your research.
 17"""
 18
 19import os
 20from glob import glob
 21from tqdm import tqdm
 22from pathlib import Path
 23from natsort import natsorted
 24from typing import Union, Tuple, List
 25
 26import numpy as np
 27import imageio.v3 as imageio
 28
 29from torch.utils.data import Dataset, DataLoader
 30
 31import torch_em
 32
 33from .. import util
 34
 35
 36KAGGLE_DATASET_NAME = "orvile/edd2020-endoscopy-detection-and-segmentation"
 37
 38CLASS_NAMES = ["BE", "suspicious", "HGD", "cancer", "polyp"]
 39
 40
 41def get_edd2020_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 42    """Download the EDD2020 dataset.
 43
 44    Args:
 45        path: Filepath to a folder where the data is downloaded for further processing.
 46        download: Whether to download the data if it is not present.
 47
 48    Returns:
 49        Filepath where the data is downloaded.
 50    """
 51    data_dir = os.path.join(path, "EDD2020")
 52    if os.path.exists(data_dir):
 53        return data_dir
 54
 55    os.makedirs(path, exist_ok=True)
 56
 57    util.download_source_kaggle(path=path, dataset_name=KAGGLE_DATASET_NAME, download=download)
 58
 59    zip_path = os.path.join(path, "edd2020-endoscopy-detection-and-segmentation.zip")
 60    util.unzip(zip_path=zip_path, dst=path)
 61
 62    if not os.path.exists(data_dir):
 63        raise RuntimeError(f"The dataset could not be found at '{path}' after extraction.")
 64
 65    return data_dir
 66
 67
 68def get_edd2020_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 69    """Get paths to the EDD2020 data.
 70
 71    Args:
 72        path: Filepath to a folder where the data is downloaded for further processing.
 73        download: Whether to download the data if it is not present.
 74
 75    Returns:
 76        List of filepaths for the image data.
 77        List of filepaths for the label data.
 78    """
 79    data_dir = get_edd2020_data(path=path, download=download)
 80
 81    image_paths = natsorted(glob(os.path.join(data_dir, "originalImages", "*.jpg")))
 82    if len(image_paths) == 0:
 83        raise RuntimeError("Something went wrong with fetching the image paths.")
 84
 85    neu_gt_dir = os.path.join(data_dir, "masks", "preprocessed")
 86    os.makedirs(neu_gt_dir, exist_ok=True)
 87
 88    gt_paths = []
 89    for image_path in tqdm(image_paths, desc="Preprocessing labels"):
 90        image_id = Path(image_path).stem
 91        gt_path = os.path.join(neu_gt_dir, f"{image_id}.tif")
 92        gt_paths.append(gt_path)
 93        if os.path.exists(gt_path):
 94            continue
 95
 96        image = imageio.imread(image_path)
 97        label = np.zeros(image.shape[:2], dtype="uint8")
 98        for class_id, class_name in enumerate(CLASS_NAMES, start=1):
 99            mask_path = os.path.join(data_dir, "masks", f"{image_id}_{class_name}.tif")
100            if not os.path.exists(mask_path):
101                continue
102            mask = imageio.imread(mask_path)
103            label[mask > 0] = class_id
104
105        imageio.imwrite(gt_path, label, compression="zlib")
106
107    return image_paths, gt_paths
108
109
110def get_edd2020_dataset(
111    path: Union[os.PathLike, str],
112    patch_shape: Tuple[int, int],
113    resize_inputs: bool = False,
114    download: bool = False,
115    **kwargs
116) -> Dataset:
117    """Get the EDD2020 dataset for multi-class disease segmentation.
118
119    Args:
120        path: Filepath to a folder where the data is downloaded for further processing.
121        patch_shape: The patch shape to use for training.
122        resize_inputs: Whether to resize the inputs to the patch shape.
123        download: Whether to download the data if it is not present.
124        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
125
126    Returns:
127        The segmentation dataset.
128    """
129    image_paths, gt_paths = get_edd2020_paths(path, download)
130
131    if resize_inputs:
132        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
133        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
134            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
135        )
136
137    return torch_em.default_segmentation_dataset(
138        raw_paths=image_paths,
139        raw_key=None,
140        label_paths=gt_paths,
141        label_key=None,
142        patch_shape=patch_shape,
143        is_seg_dataset=False,
144        **kwargs
145    )
146
147
148def get_edd2020_loader(
149    path: Union[os.PathLike, str],
150    batch_size: int,
151    patch_shape: Tuple[int, int],
152    resize_inputs: bool = False,
153    download: bool = False,
154    **kwargs
155) -> DataLoader:
156    """Get the EDD2020 dataloader for multi-class disease segmentation.
157
158    Args:
159        path: Filepath to a folder where the data is downloaded for further processing.
160        batch_size: The batch size for training.
161        patch_shape: The patch shape to use for training.
162        resize_inputs: Whether to resize the inputs to the patch shape.
163        download: Whether to download the data if it is not present.
164        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
165
166    Returns:
167        The DataLoader.
168    """
169    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
170    dataset = get_edd2020_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
171    return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
KAGGLE_DATASET_NAME = 'orvile/edd2020-endoscopy-detection-and-segmentation'
CLASS_NAMES = ['BE', 'suspicious', 'HGD', 'cancer', 'polyp']
def get_edd2020_data(path: Union[os.PathLike, str], download: bool = False) -> str:
42def get_edd2020_data(path: Union[os.PathLike, str], download: bool = False) -> str:
43    """Download the EDD2020 dataset.
44
45    Args:
46        path: Filepath to a folder where the data is downloaded for further processing.
47        download: Whether to download the data if it is not present.
48
49    Returns:
50        Filepath where the data is downloaded.
51    """
52    data_dir = os.path.join(path, "EDD2020")
53    if os.path.exists(data_dir):
54        return data_dir
55
56    os.makedirs(path, exist_ok=True)
57
58    util.download_source_kaggle(path=path, dataset_name=KAGGLE_DATASET_NAME, download=download)
59
60    zip_path = os.path.join(path, "edd2020-endoscopy-detection-and-segmentation.zip")
61    util.unzip(zip_path=zip_path, dst=path)
62
63    if not os.path.exists(data_dir):
64        raise RuntimeError(f"The dataset could not be found at '{path}' after extraction.")
65
66    return data_dir

Download the EDD2020 dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_edd2020_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 69def get_edd2020_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 70    """Get paths to the EDD2020 data.
 71
 72    Args:
 73        path: Filepath to a folder where the data is downloaded for further processing.
 74        download: Whether to download the data if it is not present.
 75
 76    Returns:
 77        List of filepaths for the image data.
 78        List of filepaths for the label data.
 79    """
 80    data_dir = get_edd2020_data(path=path, download=download)
 81
 82    image_paths = natsorted(glob(os.path.join(data_dir, "originalImages", "*.jpg")))
 83    if len(image_paths) == 0:
 84        raise RuntimeError("Something went wrong with fetching the image paths.")
 85
 86    neu_gt_dir = os.path.join(data_dir, "masks", "preprocessed")
 87    os.makedirs(neu_gt_dir, exist_ok=True)
 88
 89    gt_paths = []
 90    for image_path in tqdm(image_paths, desc="Preprocessing labels"):
 91        image_id = Path(image_path).stem
 92        gt_path = os.path.join(neu_gt_dir, f"{image_id}.tif")
 93        gt_paths.append(gt_path)
 94        if os.path.exists(gt_path):
 95            continue
 96
 97        image = imageio.imread(image_path)
 98        label = np.zeros(image.shape[:2], dtype="uint8")
 99        for class_id, class_name in enumerate(CLASS_NAMES, start=1):
100            mask_path = os.path.join(data_dir, "masks", f"{image_id}_{class_name}.tif")
101            if not os.path.exists(mask_path):
102                continue
103            mask = imageio.imread(mask_path)
104            label[mask > 0] = class_id
105
106        imageio.imwrite(gt_path, label, compression="zlib")
107
108    return image_paths, gt_paths

Get paths to the EDD2020 data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_edd2020_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
111def get_edd2020_dataset(
112    path: Union[os.PathLike, str],
113    patch_shape: Tuple[int, int],
114    resize_inputs: bool = False,
115    download: bool = False,
116    **kwargs
117) -> Dataset:
118    """Get the EDD2020 dataset for multi-class disease segmentation.
119
120    Args:
121        path: Filepath to a folder where the data is downloaded for further processing.
122        patch_shape: The patch shape to use for training.
123        resize_inputs: Whether to resize the inputs to the patch shape.
124        download: Whether to download the data if it is not present.
125        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
126
127    Returns:
128        The segmentation dataset.
129    """
130    image_paths, gt_paths = get_edd2020_paths(path, download)
131
132    if resize_inputs:
133        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
134        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
135            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
136        )
137
138    return torch_em.default_segmentation_dataset(
139        raw_paths=image_paths,
140        raw_key=None,
141        label_paths=gt_paths,
142        label_key=None,
143        patch_shape=patch_shape,
144        is_seg_dataset=False,
145        **kwargs
146    )

Get the EDD2020 dataset for multi-class disease segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_edd2020_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
149def get_edd2020_loader(
150    path: Union[os.PathLike, str],
151    batch_size: int,
152    patch_shape: Tuple[int, int],
153    resize_inputs: bool = False,
154    download: bool = False,
155    **kwargs
156) -> DataLoader:
157    """Get the EDD2020 dataloader for multi-class disease segmentation.
158
159    Args:
160        path: Filepath to a folder where the data is downloaded for further processing.
161        batch_size: The batch size for training.
162        patch_shape: The patch shape to use for training.
163        resize_inputs: Whether to resize the inputs to the patch shape.
164        download: Whether to download the data if it is not present.
165        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
166
167    Returns:
168        The DataLoader.
169    """
170    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
171    dataset = get_edd2020_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
172    return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)

Get the EDD2020 dataloader for multi-class disease segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.