torch_em.data.datasets.medical.endoscapes2023

The Endoscapes2023 dataset contains annotations for segmentation of hepatocystic anatomy and surgical tools in laparoscopic cholecystectomy images.

This is the segmentation subset (Endoscapes-Seg50) of the Endoscapes2023 dataset: 493 frames from 50 videos, annotated with semantic and instance segmentation masks for 6 anatomical structures / a tool class. The full dataset additionally contains bounding box (Endoscapes-BBox201) and Critical View of Safety (CVS) classification annotations (Endoscapes-CVS201) for many more frames, but no pixel-level masks for those; they are not covered by this module.

The dataset is located at https://github.com/CAMMA-public/Endoscapes and downloaded from https://s3.unistra.fr/camma_public/datasets/endoscapes/endoscapes.zip (~6.3 GB; the segmentation subset cannot be downloaded on its own, the full archive is always fetched). It is licensed under CC BY-NC-SA 4.0 (non-commercial research use only).

This dataset is from the publication https://doi.org/10.48550/arXiv.2312.12429. Please additionally cite https://doi.org/10.48550/arXiv.2112.13815 if you use the segmentation annotations (Endoscapes-Seg50) in a publication.

  1"""The Endoscapes2023 dataset contains annotations for segmentation of hepatocystic anatomy and
  2surgical tools in laparoscopic cholecystectomy images.
  3
  4This is the segmentation subset (Endoscapes-Seg50) of the Endoscapes2023 dataset: 493 frames from
  550 videos, annotated with semantic and instance segmentation masks for 6 anatomical structures / a
  6tool class. The full dataset additionally contains bounding box (Endoscapes-BBox201) and Critical
  7View of Safety (CVS) classification annotations (Endoscapes-CVS201) for many more frames, but no
  8pixel-level masks for those; they are not covered by this module.
  9
 10The dataset is located at https://github.com/CAMMA-public/Endoscapes and downloaded from
 11https://s3.unistra.fr/camma_public/datasets/endoscapes/endoscapes.zip (~6.3 GB; the segmentation
 12subset cannot be downloaded on its own, the full archive is always fetched). It is licensed under
 13CC BY-NC-SA 4.0 (non-commercial research use only).
 14
 15This dataset is from the publication https://doi.org/10.48550/arXiv.2312.12429.
 16Please additionally cite https://doi.org/10.48550/arXiv.2112.13815 if you use the segmentation
 17annotations (Endoscapes-Seg50) in a publication.
 18"""
 19
 20import os
 21import shutil
 22from glob import glob
 23from pathlib import Path
 24from natsort import natsorted
 25from typing import Union, Tuple, List, Literal
 26
 27import imageio.v3 as imageio
 28
 29from torch.utils.data import Dataset, DataLoader
 30
 31import torch_em
 32
 33from .. import util
 34
 35
 36URL = "https://s3.unistra.fr/camma_public/datasets/endoscapes/endoscapes.zip"
 37CHECKSUM = "0574b11a82779a1a0a783e2083627b1140de096e8b7172903e35afb583d75862"
 38
 39CLASSES = ["background", "cystic_plate", "calot_triangle", "cystic_artery", "cystic_duct", "gallbladder", "tool"]
 40"""The classes of the Endoscapes2023 segmentation masks, in order of their label id (see `seg_label_map.txt`
 41in the downloaded data)."""
 42
 43SPLIT_DIRS = {"train": "train_seg", "val": "val_seg", "test": "test_seg"}
 44
 45
 46def get_endoscapes2023_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 47    """Download the Endoscapes2023 dataset.
 48
 49    Args:
 50        path: Filepath to a folder where the data is downloaded for further processing.
 51        download: Whether to download the data if it is not present.
 52
 53    Returns:
 54        Filepath where the data is downloaded.
 55    """
 56    data_dir = os.path.join(path, "endoscapes")
 57    if os.path.exists(data_dir):
 58        return data_dir
 59
 60    os.makedirs(path, exist_ok=True)
 61
 62    zip_path = os.path.join(path, "endoscapes.zip")
 63    print("Downloading the Endoscapes2023 data. This is a ~6.3 GB archive, it might take a while.")
 64    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 65    util.unzip(zip_path=zip_path, dst=path)
 66
 67    return data_dir
 68
 69
 70def get_endoscapes2023_paths(
 71    path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False
 72) -> Tuple[List[str], List[str]]:
 73    """Get paths to the Endoscapes2023 data.
 74
 75    Args:
 76        path: Filepath to a folder where the data is downloaded for further processing.
 77        split: The choice of data split.
 78        download: Whether to download the data if it is not present.
 79
 80    Returns:
 81        List of filepaths for the image data.
 82        List of filepaths for the label data.
 83    """
 84    if split not in SPLIT_DIRS:
 85        raise ValueError(f"'{split}' is not a valid split.")
 86
 87    data_dir = get_endoscapes2023_data(path, download)
 88    image_dir = os.path.join(data_dir, SPLIT_DIRS[split])
 89
 90    ppdir = os.path.join(data_dir, "preprocessed", split)
 91    prepped_images = os.path.join(ppdir, "images")
 92    prepped_masks = os.path.join(ppdir, "masks")
 93    if os.path.exists(prepped_images) and os.path.exists(prepped_masks):
 94        return natsorted(glob(os.path.join(prepped_images, "*.jpg"))), natsorted(glob(os.path.join(prepped_masks, "*.tif")))  # noqa
 95
 96    os.makedirs(prepped_images, exist_ok=True)
 97    os.makedirs(prepped_masks, exist_ok=True)
 98
 99    # The semantic masks live in one shared 'semseg' folder for all splits; only the ones whose
100    # frame id also exists in this split's image folder belong to this split.
101    mask_paths = natsorted(glob(os.path.join(data_dir, "semseg", "*.png")))
102
103    image_paths, gt_paths = [], []
104    for mask_path in mask_paths:
105        frame_id = Path(mask_path).stem
106        src_image_path = os.path.join(image_dir, f"{frame_id}.jpg")
107        if not os.path.exists(src_image_path):
108            continue  # This mask belongs to a different split.
109
110        dst_image_path = os.path.join(prepped_images, f"{frame_id}.jpg")
111        dst_mask_path = os.path.join(prepped_masks, f"{frame_id}.tif")
112
113        image_paths.append(dst_image_path)
114        gt_paths.append(dst_mask_path)
115
116        if os.path.exists(dst_image_path) and os.path.exists(dst_mask_path):
117            continue
118
119        mask = imageio.imread(mask_path)
120        # Clean up rare annotation artifacts found in the raw masks: pixel value 255 marks
121        # unlabeled / ambiguous regions (present in about a quarter of the masks), and a single
122        # mask has a stray value of 7, which is outside the valid [0, 6] label range. Both are
123        # mapped back to the background class.
124        mask[(mask == 255) | (mask > (len(CLASSES) - 1))] = 0
125
126        shutil.copy(src_image_path, dst_image_path)
127        imageio.imwrite(dst_mask_path, mask, compression="zlib")
128
129    assert image_paths and len(image_paths) == len(gt_paths)
130    return image_paths, gt_paths
131
132
133def get_endoscapes2023_dataset(
134    path: Union[os.PathLike, str],
135    patch_shape: Tuple[int, int],
136    split: Literal["train", "val", "test"],
137    resize_inputs: bool = False,
138    download: bool = False,
139    **kwargs
140) -> Dataset:
141    """Get the Endoscapes2023 dataset for anatomy and tool segmentation in laparoscopic cholecystectomy images.
142
143    Args:
144        path: Filepath to a folder where the data is downloaded for further processing.
145        patch_shape: The patch shape to use for training.
146        split: The choice of data split.
147        resize_inputs: Whether to resize inputs to the desired patch shape.
148        download: Whether to download the data if it is not present.
149        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
150
151    Returns:
152        The segmentation dataset.
153    """
154    image_paths, gt_paths = get_endoscapes2023_paths(path, split, download)
155
156    if resize_inputs:
157        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
158        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
159            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
160        )
161
162    return torch_em.default_segmentation_dataset(
163        raw_paths=image_paths,
164        raw_key=None,
165        label_paths=gt_paths,
166        label_key=None,
167        patch_shape=patch_shape,
168        is_seg_dataset=False,
169        **kwargs
170    )
171
172
173def get_endoscapes2023_loader(
174    path: Union[os.PathLike, str],
175    batch_size: int,
176    patch_shape: Tuple[int, int],
177    split: Literal["train", "val", "test"],
178    resize_inputs: bool = False,
179    download: bool = False,
180    **kwargs
181) -> DataLoader:
182    """Get the Endoscapes2023 dataloader for anatomy and tool segmentation in laparoscopic cholecystectomy images.
183
184    Args:
185        path: Filepath to a folder where the data is downloaded for further processing.
186        batch_size: The batch size for training.
187        patch_shape: The patch shape to use for training.
188        split: The choice of data split.
189        resize_inputs: Whether to resize inputs to the desired patch shape.
190        download: Whether to download the data if it is not present.
191        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
192
193    Returns:
194        The DataLoader.
195    """
196    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
197    dataset = get_endoscapes2023_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs)
198    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://s3.unistra.fr/camma_public/datasets/endoscapes/endoscapes.zip'
CHECKSUM = '0574b11a82779a1a0a783e2083627b1140de096e8b7172903e35afb583d75862'
CLASSES = ['background', 'cystic_plate', 'calot_triangle', 'cystic_artery', 'cystic_duct', 'gallbladder', 'tool']

The classes of the Endoscapes2023 segmentation masks, in order of their label id (see seg_label_map.txt in the downloaded data).

SPLIT_DIRS = {'train': 'train_seg', 'val': 'val_seg', 'test': 'test_seg'}
def get_endoscapes2023_data(path: Union[os.PathLike, str], download: bool = False) -> str:
47def get_endoscapes2023_data(path: Union[os.PathLike, str], download: bool = False) -> str:
48    """Download the Endoscapes2023 dataset.
49
50    Args:
51        path: Filepath to a folder where the data is downloaded for further processing.
52        download: Whether to download the data if it is not present.
53
54    Returns:
55        Filepath where the data is downloaded.
56    """
57    data_dir = os.path.join(path, "endoscapes")
58    if os.path.exists(data_dir):
59        return data_dir
60
61    os.makedirs(path, exist_ok=True)
62
63    zip_path = os.path.join(path, "endoscapes.zip")
64    print("Downloading the Endoscapes2023 data. This is a ~6.3 GB archive, it might take a while.")
65    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
66    util.unzip(zip_path=zip_path, dst=path)
67
68    return data_dir

Download the Endoscapes2023 dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_endoscapes2023_paths( path: Union[os.PathLike, str], split: Literal['train', 'val', 'test'], download: bool = False) -> Tuple[List[str], List[str]]:
 71def get_endoscapes2023_paths(
 72    path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False
 73) -> Tuple[List[str], List[str]]:
 74    """Get paths to the Endoscapes2023 data.
 75
 76    Args:
 77        path: Filepath to a folder where the data is downloaded for further processing.
 78        split: The choice of data split.
 79        download: Whether to download the data if it is not present.
 80
 81    Returns:
 82        List of filepaths for the image data.
 83        List of filepaths for the label data.
 84    """
 85    if split not in SPLIT_DIRS:
 86        raise ValueError(f"'{split}' is not a valid split.")
 87
 88    data_dir = get_endoscapes2023_data(path, download)
 89    image_dir = os.path.join(data_dir, SPLIT_DIRS[split])
 90
 91    ppdir = os.path.join(data_dir, "preprocessed", split)
 92    prepped_images = os.path.join(ppdir, "images")
 93    prepped_masks = os.path.join(ppdir, "masks")
 94    if os.path.exists(prepped_images) and os.path.exists(prepped_masks):
 95        return natsorted(glob(os.path.join(prepped_images, "*.jpg"))), natsorted(glob(os.path.join(prepped_masks, "*.tif")))  # noqa
 96
 97    os.makedirs(prepped_images, exist_ok=True)
 98    os.makedirs(prepped_masks, exist_ok=True)
 99
100    # The semantic masks live in one shared 'semseg' folder for all splits; only the ones whose
101    # frame id also exists in this split's image folder belong to this split.
102    mask_paths = natsorted(glob(os.path.join(data_dir, "semseg", "*.png")))
103
104    image_paths, gt_paths = [], []
105    for mask_path in mask_paths:
106        frame_id = Path(mask_path).stem
107        src_image_path = os.path.join(image_dir, f"{frame_id}.jpg")
108        if not os.path.exists(src_image_path):
109            continue  # This mask belongs to a different split.
110
111        dst_image_path = os.path.join(prepped_images, f"{frame_id}.jpg")
112        dst_mask_path = os.path.join(prepped_masks, f"{frame_id}.tif")
113
114        image_paths.append(dst_image_path)
115        gt_paths.append(dst_mask_path)
116
117        if os.path.exists(dst_image_path) and os.path.exists(dst_mask_path):
118            continue
119
120        mask = imageio.imread(mask_path)
121        # Clean up rare annotation artifacts found in the raw masks: pixel value 255 marks
122        # unlabeled / ambiguous regions (present in about a quarter of the masks), and a single
123        # mask has a stray value of 7, which is outside the valid [0, 6] label range. Both are
124        # mapped back to the background class.
125        mask[(mask == 255) | (mask > (len(CLASSES) - 1))] = 0
126
127        shutil.copy(src_image_path, dst_image_path)
128        imageio.imwrite(dst_mask_path, mask, compression="zlib")
129
130    assert image_paths and len(image_paths) == len(gt_paths)
131    return image_paths, gt_paths

Get paths to the Endoscapes2023 data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The choice of data split.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_endoscapes2023_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], split: Literal['train', 'val', 'test'], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
134def get_endoscapes2023_dataset(
135    path: Union[os.PathLike, str],
136    patch_shape: Tuple[int, int],
137    split: Literal["train", "val", "test"],
138    resize_inputs: bool = False,
139    download: bool = False,
140    **kwargs
141) -> Dataset:
142    """Get the Endoscapes2023 dataset for anatomy and tool segmentation in laparoscopic cholecystectomy images.
143
144    Args:
145        path: Filepath to a folder where the data is downloaded for further processing.
146        patch_shape: The patch shape to use for training.
147        split: The choice of data split.
148        resize_inputs: Whether to resize inputs to the desired patch shape.
149        download: Whether to download the data if it is not present.
150        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
151
152    Returns:
153        The segmentation dataset.
154    """
155    image_paths, gt_paths = get_endoscapes2023_paths(path, split, download)
156
157    if resize_inputs:
158        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
159        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
160            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
161        )
162
163    return torch_em.default_segmentation_dataset(
164        raw_paths=image_paths,
165        raw_key=None,
166        label_paths=gt_paths,
167        label_key=None,
168        patch_shape=patch_shape,
169        is_seg_dataset=False,
170        **kwargs
171    )

Get the Endoscapes2023 dataset for anatomy and tool segmentation in laparoscopic cholecystectomy images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_endoscapes2023_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], split: Literal['train', 'val', 'test'], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
174def get_endoscapes2023_loader(
175    path: Union[os.PathLike, str],
176    batch_size: int,
177    patch_shape: Tuple[int, int],
178    split: Literal["train", "val", "test"],
179    resize_inputs: bool = False,
180    download: bool = False,
181    **kwargs
182) -> DataLoader:
183    """Get the Endoscapes2023 dataloader for anatomy and tool segmentation in laparoscopic cholecystectomy images.
184
185    Args:
186        path: Filepath to a folder where the data is downloaded for further processing.
187        batch_size: The batch size for training.
188        patch_shape: The patch shape to use for training.
189        split: The choice of data split.
190        resize_inputs: Whether to resize inputs to the desired patch shape.
191        download: Whether to download the data if it is not present.
192        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
193
194    Returns:
195        The DataLoader.
196    """
197    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
198    dataset = get_endoscapes2023_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs)
199    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the Endoscapes2023 dataloader for anatomy and tool segmentation in laparoscopic cholecystectomy images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.