torch_em.data.datasets.medical.edd2020
The EDD2020 dataset contains annotations for multi-class disease segmentation in gastrointestinal endoscopy images.
The dataset consists of 386 endoscopy frames from 5 different institutions and multiple GI organs (esophagus, stomach, colon), collected with white light, narrow-band imaging and chromoendoscopy. Each image is annotated with pixel-level masks for up to 5 disease classes: non-dysplastic Barrett's esophagus (BE), suspicious lesions, high-grade dysplasia (HGD), cancer, and polyp. The official challenge data is gated behind manual registration on https://edd2020.grand-challenge.org, so we instead use the openly mirrored copy at https://www.kaggle.com/datasets/orvile/edd2020-endoscopy-detection-and-segmentation, which matches the official release (386 images, same organizers and class structure). NOTE: the original data release states the license as CC BY-NC-SA 4.0, while the Kaggle mirror lists it as CC BY 4.0; please check the current license terms before using this data.
This dataset is from the publication https://doi.org/10.1016/j.media.2021.102002. Please cite it if you use this dataset for your research.
1"""The EDD2020 dataset contains annotations for multi-class disease segmentation in 2gastrointestinal endoscopy images. 3 4The dataset consists of 386 endoscopy frames from 5 different institutions and multiple 5GI organs (esophagus, stomach, colon), collected with white light, narrow-band imaging and 6chromoendoscopy. Each image is annotated with pixel-level masks for up to 5 disease classes: 7non-dysplastic Barrett's esophagus (BE), suspicious lesions, high-grade dysplasia (HGD), 8cancer, and polyp. The official challenge data is gated behind manual registration on 9https://edd2020.grand-challenge.org, so we instead use the openly mirrored copy at 10https://www.kaggle.com/datasets/orvile/edd2020-endoscopy-detection-and-segmentation, which 11matches the official release (386 images, same organizers and class structure). NOTE: the 12original data release states the license as CC BY-NC-SA 4.0, while the Kaggle mirror lists it 13as CC BY 4.0; please check the current license terms before using this data. 14 15This dataset is from the publication https://doi.org/10.1016/j.media.2021.102002. 16Please cite it if you use this dataset for your research. 17""" 18 19import os 20from glob import glob 21from tqdm import tqdm 22from pathlib import Path 23from natsort import natsorted 24from typing import Union, Tuple, List 25 26import numpy as np 27import imageio.v3 as imageio 28 29from torch.utils.data import Dataset, DataLoader 30 31import torch_em 32 33from .. import util 34 35 36KAGGLE_DATASET_NAME = "orvile/edd2020-endoscopy-detection-and-segmentation" 37 38CLASS_NAMES = ["BE", "suspicious", "HGD", "cancer", "polyp"] 39 40 41def get_edd2020_data(path: Union[os.PathLike, str], download: bool = False) -> str: 42 """Download the EDD2020 dataset. 43 44 Args: 45 path: Filepath to a folder where the data is downloaded for further processing. 46 download: Whether to download the data if it is not present. 47 48 Returns: 49 Filepath where the data is downloaded. 50 """ 51 data_dir = os.path.join(path, "EDD2020") 52 if os.path.exists(data_dir): 53 return data_dir 54 55 os.makedirs(path, exist_ok=True) 56 57 util.download_source_kaggle(path=path, dataset_name=KAGGLE_DATASET_NAME, download=download) 58 59 zip_path = os.path.join(path, "edd2020-endoscopy-detection-and-segmentation.zip") 60 util.unzip(zip_path=zip_path, dst=path) 61 62 if not os.path.exists(data_dir): 63 raise RuntimeError(f"The dataset could not be found at '{path}' after extraction.") 64 65 return data_dir 66 67 68def get_edd2020_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 69 """Get paths to the EDD2020 data. 70 71 Args: 72 path: Filepath to a folder where the data is downloaded for further processing. 73 download: Whether to download the data if it is not present. 74 75 Returns: 76 List of filepaths for the image data. 77 List of filepaths for the label data. 78 """ 79 data_dir = get_edd2020_data(path=path, download=download) 80 81 image_paths = natsorted(glob(os.path.join(data_dir, "originalImages", "*.jpg"))) 82 if len(image_paths) == 0: 83 raise RuntimeError("Something went wrong with fetching the image paths.") 84 85 neu_gt_dir = os.path.join(data_dir, "masks", "preprocessed") 86 os.makedirs(neu_gt_dir, exist_ok=True) 87 88 gt_paths = [] 89 for image_path in tqdm(image_paths, desc="Preprocessing labels"): 90 image_id = Path(image_path).stem 91 gt_path = os.path.join(neu_gt_dir, f"{image_id}.tif") 92 gt_paths.append(gt_path) 93 if os.path.exists(gt_path): 94 continue 95 96 image = imageio.imread(image_path) 97 label = np.zeros(image.shape[:2], dtype="uint8") 98 for class_id, class_name in enumerate(CLASS_NAMES, start=1): 99 mask_path = os.path.join(data_dir, "masks", f"{image_id}_{class_name}.tif") 100 if not os.path.exists(mask_path): 101 continue 102 mask = imageio.imread(mask_path) 103 label[mask > 0] = class_id 104 105 imageio.imwrite(gt_path, label, compression="zlib") 106 107 return image_paths, gt_paths 108 109 110def get_edd2020_dataset( 111 path: Union[os.PathLike, str], 112 patch_shape: Tuple[int, int], 113 resize_inputs: bool = False, 114 download: bool = False, 115 **kwargs 116) -> Dataset: 117 """Get the EDD2020 dataset for multi-class disease segmentation. 118 119 Args: 120 path: Filepath to a folder where the data is downloaded for further processing. 121 patch_shape: The patch shape to use for training. 122 resize_inputs: Whether to resize the inputs to the patch shape. 123 download: Whether to download the data if it is not present. 124 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 125 126 Returns: 127 The segmentation dataset. 128 """ 129 image_paths, gt_paths = get_edd2020_paths(path, download) 130 131 if resize_inputs: 132 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 133 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 134 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 135 ) 136 137 return torch_em.default_segmentation_dataset( 138 raw_paths=image_paths, 139 raw_key=None, 140 label_paths=gt_paths, 141 label_key=None, 142 patch_shape=patch_shape, 143 is_seg_dataset=False, 144 **kwargs 145 ) 146 147 148def get_edd2020_loader( 149 path: Union[os.PathLike, str], 150 batch_size: int, 151 patch_shape: Tuple[int, int], 152 resize_inputs: bool = False, 153 download: bool = False, 154 **kwargs 155) -> DataLoader: 156 """Get the EDD2020 dataloader for multi-class disease segmentation. 157 158 Args: 159 path: Filepath to a folder where the data is downloaded for further processing. 160 batch_size: The batch size for training. 161 patch_shape: The patch shape to use for training. 162 resize_inputs: Whether to resize the inputs to the patch shape. 163 download: Whether to download the data if it is not present. 164 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 165 166 Returns: 167 The DataLoader. 168 """ 169 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 170 dataset = get_edd2020_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 171 return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
42def get_edd2020_data(path: Union[os.PathLike, str], download: bool = False) -> str: 43 """Download the EDD2020 dataset. 44 45 Args: 46 path: Filepath to a folder where the data is downloaded for further processing. 47 download: Whether to download the data if it is not present. 48 49 Returns: 50 Filepath where the data is downloaded. 51 """ 52 data_dir = os.path.join(path, "EDD2020") 53 if os.path.exists(data_dir): 54 return data_dir 55 56 os.makedirs(path, exist_ok=True) 57 58 util.download_source_kaggle(path=path, dataset_name=KAGGLE_DATASET_NAME, download=download) 59 60 zip_path = os.path.join(path, "edd2020-endoscopy-detection-and-segmentation.zip") 61 util.unzip(zip_path=zip_path, dst=path) 62 63 if not os.path.exists(data_dir): 64 raise RuntimeError(f"The dataset could not be found at '{path}' after extraction.") 65 66 return data_dir
Download the EDD2020 dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
69def get_edd2020_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 70 """Get paths to the EDD2020 data. 71 72 Args: 73 path: Filepath to a folder where the data is downloaded for further processing. 74 download: Whether to download the data if it is not present. 75 76 Returns: 77 List of filepaths for the image data. 78 List of filepaths for the label data. 79 """ 80 data_dir = get_edd2020_data(path=path, download=download) 81 82 image_paths = natsorted(glob(os.path.join(data_dir, "originalImages", "*.jpg"))) 83 if len(image_paths) == 0: 84 raise RuntimeError("Something went wrong with fetching the image paths.") 85 86 neu_gt_dir = os.path.join(data_dir, "masks", "preprocessed") 87 os.makedirs(neu_gt_dir, exist_ok=True) 88 89 gt_paths = [] 90 for image_path in tqdm(image_paths, desc="Preprocessing labels"): 91 image_id = Path(image_path).stem 92 gt_path = os.path.join(neu_gt_dir, f"{image_id}.tif") 93 gt_paths.append(gt_path) 94 if os.path.exists(gt_path): 95 continue 96 97 image = imageio.imread(image_path) 98 label = np.zeros(image.shape[:2], dtype="uint8") 99 for class_id, class_name in enumerate(CLASS_NAMES, start=1): 100 mask_path = os.path.join(data_dir, "masks", f"{image_id}_{class_name}.tif") 101 if not os.path.exists(mask_path): 102 continue 103 mask = imageio.imread(mask_path) 104 label[mask > 0] = class_id 105 106 imageio.imwrite(gt_path, label, compression="zlib") 107 108 return image_paths, gt_paths
Get paths to the EDD2020 data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
111def get_edd2020_dataset( 112 path: Union[os.PathLike, str], 113 patch_shape: Tuple[int, int], 114 resize_inputs: bool = False, 115 download: bool = False, 116 **kwargs 117) -> Dataset: 118 """Get the EDD2020 dataset for multi-class disease segmentation. 119 120 Args: 121 path: Filepath to a folder where the data is downloaded for further processing. 122 patch_shape: The patch shape to use for training. 123 resize_inputs: Whether to resize the inputs to the patch shape. 124 download: Whether to download the data if it is not present. 125 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 126 127 Returns: 128 The segmentation dataset. 129 """ 130 image_paths, gt_paths = get_edd2020_paths(path, download) 131 132 if resize_inputs: 133 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 134 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 135 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 136 ) 137 138 return torch_em.default_segmentation_dataset( 139 raw_paths=image_paths, 140 raw_key=None, 141 label_paths=gt_paths, 142 label_key=None, 143 patch_shape=patch_shape, 144 is_seg_dataset=False, 145 **kwargs 146 )
Get the EDD2020 dataset for multi-class disease segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
149def get_edd2020_loader( 150 path: Union[os.PathLike, str], 151 batch_size: int, 152 patch_shape: Tuple[int, int], 153 resize_inputs: bool = False, 154 download: bool = False, 155 **kwargs 156) -> DataLoader: 157 """Get the EDD2020 dataloader for multi-class disease segmentation. 158 159 Args: 160 path: Filepath to a folder where the data is downloaded for further processing. 161 batch_size: The batch size for training. 162 patch_shape: The patch shape to use for training. 163 resize_inputs: Whether to resize the inputs to the patch shape. 164 download: Whether to download the data if it is not present. 165 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 166 167 Returns: 168 The DataLoader. 169 """ 170 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 171 dataset = get_edd2020_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 172 return torch_em.get_data_loader(dataset=dataset, batch_size=batch_size, **loader_kwargs)
Get the EDD2020 dataloader for multi-class disease segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.