torch_em.data.datasets.medical.endoscapes2023
The Endoscapes2023 dataset contains annotations for segmentation of hepatocystic anatomy and surgical tools in laparoscopic cholecystectomy images.
This is the segmentation subset (Endoscapes-Seg50) of the Endoscapes2023 dataset: 493 frames from 50 videos, annotated with semantic and instance segmentation masks for 6 anatomical structures / a tool class. The full dataset additionally contains bounding box (Endoscapes-BBox201) and Critical View of Safety (CVS) classification annotations (Endoscapes-CVS201) for many more frames, but no pixel-level masks for those; they are not covered by this module.
The dataset is located at https://github.com/CAMMA-public/Endoscapes and downloaded from https://s3.unistra.fr/camma_public/datasets/endoscapes/endoscapes.zip (~6.3 GB; the segmentation subset cannot be downloaded on its own, the full archive is always fetched). It is licensed under CC BY-NC-SA 4.0 (non-commercial research use only).
This dataset is from the publication https://doi.org/10.48550/arXiv.2312.12429. Please additionally cite https://doi.org/10.48550/arXiv.2112.13815 if you use the segmentation annotations (Endoscapes-Seg50) in a publication.
1"""The Endoscapes2023 dataset contains annotations for segmentation of hepatocystic anatomy and 2surgical tools in laparoscopic cholecystectomy images. 3 4This is the segmentation subset (Endoscapes-Seg50) of the Endoscapes2023 dataset: 493 frames from 550 videos, annotated with semantic and instance segmentation masks for 6 anatomical structures / a 6tool class. The full dataset additionally contains bounding box (Endoscapes-BBox201) and Critical 7View of Safety (CVS) classification annotations (Endoscapes-CVS201) for many more frames, but no 8pixel-level masks for those; they are not covered by this module. 9 10The dataset is located at https://github.com/CAMMA-public/Endoscapes and downloaded from 11https://s3.unistra.fr/camma_public/datasets/endoscapes/endoscapes.zip (~6.3 GB; the segmentation 12subset cannot be downloaded on its own, the full archive is always fetched). It is licensed under 13CC BY-NC-SA 4.0 (non-commercial research use only). 14 15This dataset is from the publication https://doi.org/10.48550/arXiv.2312.12429. 16Please additionally cite https://doi.org/10.48550/arXiv.2112.13815 if you use the segmentation 17annotations (Endoscapes-Seg50) in a publication. 18""" 19 20import os 21import shutil 22from glob import glob 23from pathlib import Path 24from natsort import natsorted 25from typing import Union, Tuple, List, Literal 26 27import imageio.v3 as imageio 28 29from torch.utils.data import Dataset, DataLoader 30 31import torch_em 32 33from .. import util 34 35 36URL = "https://s3.unistra.fr/camma_public/datasets/endoscapes/endoscapes.zip" 37CHECKSUM = "0574b11a82779a1a0a783e2083627b1140de096e8b7172903e35afb583d75862" 38 39CLASSES = ["background", "cystic_plate", "calot_triangle", "cystic_artery", "cystic_duct", "gallbladder", "tool"] 40"""The classes of the Endoscapes2023 segmentation masks, in order of their label id (see `seg_label_map.txt` 41in the downloaded data).""" 42 43SPLIT_DIRS = {"train": "train_seg", "val": "val_seg", "test": "test_seg"} 44 45 46def get_endoscapes2023_data(path: Union[os.PathLike, str], download: bool = False) -> str: 47 """Download the Endoscapes2023 dataset. 48 49 Args: 50 path: Filepath to a folder where the data is downloaded for further processing. 51 download: Whether to download the data if it is not present. 52 53 Returns: 54 Filepath where the data is downloaded. 55 """ 56 data_dir = os.path.join(path, "endoscapes") 57 if os.path.exists(data_dir): 58 return data_dir 59 60 os.makedirs(path, exist_ok=True) 61 62 zip_path = os.path.join(path, "endoscapes.zip") 63 print("Downloading the Endoscapes2023 data. This is a ~6.3 GB archive, it might take a while.") 64 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 65 util.unzip(zip_path=zip_path, dst=path) 66 67 return data_dir 68 69 70def get_endoscapes2023_paths( 71 path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False 72) -> Tuple[List[str], List[str]]: 73 """Get paths to the Endoscapes2023 data. 74 75 Args: 76 path: Filepath to a folder where the data is downloaded for further processing. 77 split: The choice of data split. 78 download: Whether to download the data if it is not present. 79 80 Returns: 81 List of filepaths for the image data. 82 List of filepaths for the label data. 83 """ 84 if split not in SPLIT_DIRS: 85 raise ValueError(f"'{split}' is not a valid split.") 86 87 data_dir = get_endoscapes2023_data(path, download) 88 image_dir = os.path.join(data_dir, SPLIT_DIRS[split]) 89 90 ppdir = os.path.join(data_dir, "preprocessed", split) 91 prepped_images = os.path.join(ppdir, "images") 92 prepped_masks = os.path.join(ppdir, "masks") 93 if os.path.exists(prepped_images) and os.path.exists(prepped_masks): 94 return natsorted(glob(os.path.join(prepped_images, "*.jpg"))), natsorted(glob(os.path.join(prepped_masks, "*.tif"))) # noqa 95 96 os.makedirs(prepped_images, exist_ok=True) 97 os.makedirs(prepped_masks, exist_ok=True) 98 99 # The semantic masks live in one shared 'semseg' folder for all splits; only the ones whose 100 # frame id also exists in this split's image folder belong to this split. 101 mask_paths = natsorted(glob(os.path.join(data_dir, "semseg", "*.png"))) 102 103 image_paths, gt_paths = [], [] 104 for mask_path in mask_paths: 105 frame_id = Path(mask_path).stem 106 src_image_path = os.path.join(image_dir, f"{frame_id}.jpg") 107 if not os.path.exists(src_image_path): 108 continue # This mask belongs to a different split. 109 110 dst_image_path = os.path.join(prepped_images, f"{frame_id}.jpg") 111 dst_mask_path = os.path.join(prepped_masks, f"{frame_id}.tif") 112 113 image_paths.append(dst_image_path) 114 gt_paths.append(dst_mask_path) 115 116 if os.path.exists(dst_image_path) and os.path.exists(dst_mask_path): 117 continue 118 119 mask = imageio.imread(mask_path) 120 # Clean up rare annotation artifacts found in the raw masks: pixel value 255 marks 121 # unlabeled / ambiguous regions (present in about a quarter of the masks), and a single 122 # mask has a stray value of 7, which is outside the valid [0, 6] label range. Both are 123 # mapped back to the background class. 124 mask[(mask == 255) | (mask > (len(CLASSES) - 1))] = 0 125 126 shutil.copy(src_image_path, dst_image_path) 127 imageio.imwrite(dst_mask_path, mask, compression="zlib") 128 129 assert image_paths and len(image_paths) == len(gt_paths) 130 return image_paths, gt_paths 131 132 133def get_endoscapes2023_dataset( 134 path: Union[os.PathLike, str], 135 patch_shape: Tuple[int, int], 136 split: Literal["train", "val", "test"], 137 resize_inputs: bool = False, 138 download: bool = False, 139 **kwargs 140) -> Dataset: 141 """Get the Endoscapes2023 dataset for anatomy and tool segmentation in laparoscopic cholecystectomy images. 142 143 Args: 144 path: Filepath to a folder where the data is downloaded for further processing. 145 patch_shape: The patch shape to use for training. 146 split: The choice of data split. 147 resize_inputs: Whether to resize inputs to the desired patch shape. 148 download: Whether to download the data if it is not present. 149 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 150 151 Returns: 152 The segmentation dataset. 153 """ 154 image_paths, gt_paths = get_endoscapes2023_paths(path, split, download) 155 156 if resize_inputs: 157 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 158 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 159 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 160 ) 161 162 return torch_em.default_segmentation_dataset( 163 raw_paths=image_paths, 164 raw_key=None, 165 label_paths=gt_paths, 166 label_key=None, 167 patch_shape=patch_shape, 168 is_seg_dataset=False, 169 **kwargs 170 ) 171 172 173def get_endoscapes2023_loader( 174 path: Union[os.PathLike, str], 175 batch_size: int, 176 patch_shape: Tuple[int, int], 177 split: Literal["train", "val", "test"], 178 resize_inputs: bool = False, 179 download: bool = False, 180 **kwargs 181) -> DataLoader: 182 """Get the Endoscapes2023 dataloader for anatomy and tool segmentation in laparoscopic cholecystectomy images. 183 184 Args: 185 path: Filepath to a folder where the data is downloaded for further processing. 186 batch_size: The batch size for training. 187 patch_shape: The patch shape to use for training. 188 split: The choice of data split. 189 resize_inputs: Whether to resize inputs to the desired patch shape. 190 download: Whether to download the data if it is not present. 191 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 192 193 Returns: 194 The DataLoader. 195 """ 196 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 197 dataset = get_endoscapes2023_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs) 198 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
The classes of the Endoscapes2023 segmentation masks, in order of their label id (see seg_label_map.txt
in the downloaded data).
47def get_endoscapes2023_data(path: Union[os.PathLike, str], download: bool = False) -> str: 48 """Download the Endoscapes2023 dataset. 49 50 Args: 51 path: Filepath to a folder where the data is downloaded for further processing. 52 download: Whether to download the data if it is not present. 53 54 Returns: 55 Filepath where the data is downloaded. 56 """ 57 data_dir = os.path.join(path, "endoscapes") 58 if os.path.exists(data_dir): 59 return data_dir 60 61 os.makedirs(path, exist_ok=True) 62 63 zip_path = os.path.join(path, "endoscapes.zip") 64 print("Downloading the Endoscapes2023 data. This is a ~6.3 GB archive, it might take a while.") 65 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 66 util.unzip(zip_path=zip_path, dst=path) 67 68 return data_dir
Download the Endoscapes2023 dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
71def get_endoscapes2023_paths( 72 path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False 73) -> Tuple[List[str], List[str]]: 74 """Get paths to the Endoscapes2023 data. 75 76 Args: 77 path: Filepath to a folder where the data is downloaded for further processing. 78 split: The choice of data split. 79 download: Whether to download the data if it is not present. 80 81 Returns: 82 List of filepaths for the image data. 83 List of filepaths for the label data. 84 """ 85 if split not in SPLIT_DIRS: 86 raise ValueError(f"'{split}' is not a valid split.") 87 88 data_dir = get_endoscapes2023_data(path, download) 89 image_dir = os.path.join(data_dir, SPLIT_DIRS[split]) 90 91 ppdir = os.path.join(data_dir, "preprocessed", split) 92 prepped_images = os.path.join(ppdir, "images") 93 prepped_masks = os.path.join(ppdir, "masks") 94 if os.path.exists(prepped_images) and os.path.exists(prepped_masks): 95 return natsorted(glob(os.path.join(prepped_images, "*.jpg"))), natsorted(glob(os.path.join(prepped_masks, "*.tif"))) # noqa 96 97 os.makedirs(prepped_images, exist_ok=True) 98 os.makedirs(prepped_masks, exist_ok=True) 99 100 # The semantic masks live in one shared 'semseg' folder for all splits; only the ones whose 101 # frame id also exists in this split's image folder belong to this split. 102 mask_paths = natsorted(glob(os.path.join(data_dir, "semseg", "*.png"))) 103 104 image_paths, gt_paths = [], [] 105 for mask_path in mask_paths: 106 frame_id = Path(mask_path).stem 107 src_image_path = os.path.join(image_dir, f"{frame_id}.jpg") 108 if not os.path.exists(src_image_path): 109 continue # This mask belongs to a different split. 110 111 dst_image_path = os.path.join(prepped_images, f"{frame_id}.jpg") 112 dst_mask_path = os.path.join(prepped_masks, f"{frame_id}.tif") 113 114 image_paths.append(dst_image_path) 115 gt_paths.append(dst_mask_path) 116 117 if os.path.exists(dst_image_path) and os.path.exists(dst_mask_path): 118 continue 119 120 mask = imageio.imread(mask_path) 121 # Clean up rare annotation artifacts found in the raw masks: pixel value 255 marks 122 # unlabeled / ambiguous regions (present in about a quarter of the masks), and a single 123 # mask has a stray value of 7, which is outside the valid [0, 6] label range. Both are 124 # mapped back to the background class. 125 mask[(mask == 255) | (mask > (len(CLASSES) - 1))] = 0 126 127 shutil.copy(src_image_path, dst_image_path) 128 imageio.imwrite(dst_mask_path, mask, compression="zlib") 129 130 assert image_paths and len(image_paths) == len(gt_paths) 131 return image_paths, gt_paths
Get paths to the Endoscapes2023 data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- split: The choice of data split.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
134def get_endoscapes2023_dataset( 135 path: Union[os.PathLike, str], 136 patch_shape: Tuple[int, int], 137 split: Literal["train", "val", "test"], 138 resize_inputs: bool = False, 139 download: bool = False, 140 **kwargs 141) -> Dataset: 142 """Get the Endoscapes2023 dataset for anatomy and tool segmentation in laparoscopic cholecystectomy images. 143 144 Args: 145 path: Filepath to a folder where the data is downloaded for further processing. 146 patch_shape: The patch shape to use for training. 147 split: The choice of data split. 148 resize_inputs: Whether to resize inputs to the desired patch shape. 149 download: Whether to download the data if it is not present. 150 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 151 152 Returns: 153 The segmentation dataset. 154 """ 155 image_paths, gt_paths = get_endoscapes2023_paths(path, split, download) 156 157 if resize_inputs: 158 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 159 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 160 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 161 ) 162 163 return torch_em.default_segmentation_dataset( 164 raw_paths=image_paths, 165 raw_key=None, 166 label_paths=gt_paths, 167 label_key=None, 168 patch_shape=patch_shape, 169 is_seg_dataset=False, 170 **kwargs 171 )
Get the Endoscapes2023 dataset for anatomy and tool segmentation in laparoscopic cholecystectomy images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- split: The choice of data split.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
174def get_endoscapes2023_loader( 175 path: Union[os.PathLike, str], 176 batch_size: int, 177 patch_shape: Tuple[int, int], 178 split: Literal["train", "val", "test"], 179 resize_inputs: bool = False, 180 download: bool = False, 181 **kwargs 182) -> DataLoader: 183 """Get the Endoscapes2023 dataloader for anatomy and tool segmentation in laparoscopic cholecystectomy images. 184 185 Args: 186 path: Filepath to a folder where the data is downloaded for further processing. 187 batch_size: The batch size for training. 188 patch_shape: The patch shape to use for training. 189 split: The choice of data split. 190 resize_inputs: Whether to resize inputs to the desired patch shape. 191 download: Whether to download the data if it is not present. 192 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 193 194 Returns: 195 The DataLoader. 196 """ 197 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 198 dataset = get_endoscapes2023_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs) 199 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the Endoscapes2023 dataloader for anatomy and tool segmentation in laparoscopic cholecystectomy images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- split: The choice of data split.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.