torch_em.data.datasets.medical.cholec_instance_seg

CholecInstanceSeg contains instance segmentation annotations for surgical instruments in laparoscopic cholecystectomy video frames.

The full CholecInstanceSeg release has four subsets, annotated on top of the frames of four existing datasets (CholecSeg8k, T50, T80 and the CholecInstanceSeg-specific full set). This module only covers the 'Instance-CholecSeg8k' subset: 8,080 frames from 17 sequences, which is the identical frame set already integrated in 'torch_em/data/datasets/medical/cholecseg8k.py', now additionally annotated with per-instance polygons for two instrument classes ('grasper' and 'hook'). The raw frames themselves are downloaded via 'get_cholecseg8k_data' (from Kaggle); this module only downloads the instance annotations and rasterizes them into instance masks.

NOTE: The annotations are hosted on Synapse (project 'syn60239970'). The Synapse wiki for this project states the license as CC BY 4.0 (not CC BY-NC-ND, as an earlier, unverified note about this dataset had assumed - the Synapse project has no access requirements and its wiki's 'License' section links to the standard CC BY 4.0 license). Downloading it requires the 'synapseclient' python library and a Synapse account with an authentication token stored in the '~/.synapseConfig' file. See 'get_cholec_instance_seg_data' for details.

The dataset is located at https://www.synapse.org/Synapse:syn60239970. This dataset is from the publication https://doi.org/10.1038/s41597-025-05163-w. Please cite it if you use this dataset in your research.

  1"""CholecInstanceSeg contains instance segmentation annotations for surgical instruments
  2in laparoscopic cholecystectomy video frames.
  3
  4The full CholecInstanceSeg release has four subsets, annotated on top of the frames of four
  5existing datasets (CholecSeg8k, T50, T80 and the CholecInstanceSeg-specific full set). This
  6module only covers the 'Instance-CholecSeg8k' subset: 8,080 frames from 17 sequences, which
  7is the identical frame set already integrated in 'torch_em/data/datasets/medical/cholecseg8k.py',
  8now additionally annotated with per-instance polygons for two instrument classes ('grasper'
  9and 'hook'). The raw frames themselves are downloaded via 'get_cholecseg8k_data' (from Kaggle);
 10this module only downloads the instance annotations and rasterizes them into instance masks.
 11
 12NOTE: The annotations are hosted on Synapse (project 'syn60239970'). The Synapse wiki for this
 13project states the license as CC BY 4.0 (not CC BY-NC-ND, as an earlier, unverified note about
 14this dataset had assumed - the Synapse project has no access requirements and its wiki's
 15'License' section links to the standard CC BY 4.0 license). Downloading it requires the
 16'synapseclient' python library and a Synapse account with an authentication token stored in the
 17'~/.synapseConfig' file. See 'get_cholec_instance_seg_data' for details.
 18
 19The dataset is located at https://www.synapse.org/Synapse:syn60239970.
 20This dataset is from the publication https://doi.org/10.1038/s41597-025-05163-w.
 21Please cite it if you use this dataset in your research.
 22"""
 23
 24import os
 25import re
 26import json
 27from glob import glob
 28from tqdm import tqdm
 29from pathlib import Path
 30from natsort import natsorted
 31from typing import Tuple, Union, Literal, List
 32
 33import numpy as np
 34import imageio.v3 as imageio
 35from skimage.draw import polygon as sk_polygon
 36
 37from torch.utils.data import Dataset, DataLoader
 38
 39import torch_em
 40
 41from .. import util
 42from .cholecseg8k import get_cholecseg8k_data
 43
 44
 45ENTITY = "syn66477541"
 46
 47FRAME_PATTERN = re.compile(r"seg8k_video(\d+)_(\d+)\.json")
 48
 49
 50def get_cholec_instance_seg_data(path: Union[os.PathLike, str], download: bool = False) -> Tuple[str, str]:
 51    """Download the CholecInstanceSeg annotations (Instance-CholecSeg8k subset) and the raw CholecSeg8k frames.
 52
 53    Follow the instructions below to get access to the Synapse-hosted annotations.
 54    - Create a free account at https://www.synapse.org.
 55    - Generate a personal access token and store it in a '~/.synapseConfig' file, see
 56      https://python-docs.synapse.org/tutorials/authentication/ for details.
 57    - Install the 'synapseclient' python library.
 58
 59    Args:
 60        path: Filepath to a folder where the data is downloaded for further processing.
 61        download: Whether to download the data if it is not present.
 62
 63    Returns:
 64        Filepath to the extracted CholecInstanceSeg annotations.
 65        Filepath to the raw CholecSeg8k frames.
 66    """
 67    ann_dir = os.path.join(path, "cholecinstanceseg")
 68    if not os.path.exists(ann_dir):
 69        os.makedirs(path, exist_ok=True)
 70
 71        import synapseclient
 72
 73        syn = synapseclient.Synapse()
 74        syn.login()
 75        zip_path = os.path.join(path, "cholecinstanceseg.zip")
 76        if not os.path.exists(zip_path):
 77            if not download:
 78                raise RuntimeError(f"Cannot find the data at {zip_path}, but download was set to False.")
 79            syn.get(ENTITY, downloadLocation=path, downloadFile=True)
 80
 81        util.unzip(zip_path=zip_path, dst=path, remove=False)
 82
 83    raw_data_dir = get_cholecseg8k_data(path, download)
 84
 85    return ann_dir, raw_data_dir
 86
 87
 88def _rasterize_instances(annotation, shape):
 89    instances = np.zeros(shape, dtype="uint16")
 90    for i, shape_ann in enumerate(annotation["shapes"], start=1):
 91        points = np.array(shape_ann["points"])
 92        rr, cc = sk_polygon(points[:, 1], points[:, 0], shape)
 93        instances[rr, cc] = i
 94    return instances
 95
 96
 97def get_cholec_instance_seg_paths(
 98    path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False
 99) -> Tuple[List[str], List[str]]:
100    """Get paths for the CholecInstanceSeg (Instance-CholecSeg8k subset) dataset.
101
102    Args:
103        path: Filepath to a folder where the data is downloaded for further processing.
104        split: The choice of data split.
105        download: Whether to download the data if it is not present.
106
107    Returns:
108        List of filepaths for the image data.
109        List of filepaths for the label data.
110    """
111    if split not in ["train", "val", "test"]:
112        raise ValueError(f"'{split}' is not a valid split. Please choose from 'train', 'val' or 'test'.")
113
114    ann_dir, raw_data_dir = get_cholec_instance_seg_data(path, download)
115
116    json_paths = natsorted(glob(os.path.join(ann_dir, split, "VID*_seg8k", "ann_dir", "*.json")))
117    assert len(json_paths) > 0, f"No annotations were found at '{os.path.join(ann_dir, split)}'."
118
119    ppdir = os.path.join(ann_dir, "preprocessed", split)
120    os.makedirs(os.path.join(ppdir, "images"), exist_ok=True)
121    os.makedirs(os.path.join(ppdir, "masks"), exist_ok=True)
122
123    image_paths, gt_paths = [], []
124    for json_path in tqdm(json_paths, desc=f"Preprocessing CholecInstanceSeg '{split}' split"):
125        match = FRAME_PATTERN.match(os.path.basename(json_path))
126        assert match is not None, f"Unexpected annotation filename: '{json_path}'."
127        video_id, frame_id = match.group(1), int(match.group(2))
128
129        org_image_paths = glob(
130            os.path.join(raw_data_dir, f"video{int(video_id):02d}", "video*", f"frame_{frame_id}_endo.png")
131        )
132        assert len(org_image_paths) == 1, (
133            f"Expected exactly one matching CholecSeg8k frame for video {video_id}, frame {frame_id}, "
134            f"found {len(org_image_paths)}."
135        )
136        org_image_path = org_image_paths[0]
137
138        image_id = os.path.basename(org_image_path)
139        image_path = os.path.join(ppdir, "images", image_id)
140        gt_path = os.path.join(ppdir, "masks", Path(image_id).with_suffix(".tif"))
141
142        image_paths.append(image_path)
143        gt_paths.append(gt_path)
144
145        if os.path.exists(image_path) and os.path.exists(gt_path):
146            continue
147
148        if not os.path.exists(image_path):
149            os.symlink(os.path.abspath(org_image_path), image_path)
150
151        with open(json_path) as f:
152            annotation = json.load(f)
153
154        # Frames without any annotated instrument instances lack the 'imageHeight' / 'imageWidth' keys,
155        # so the raw image shape is used as a fallback for the rasterized mask shape.
156        if "imageHeight" in annotation and "imageWidth" in annotation:
157            shape = (annotation["imageHeight"], annotation["imageWidth"])
158        else:
159            shape = imageio.imread(org_image_path).shape[:2]
160
161        instances = _rasterize_instances(annotation, shape)
162        imageio.imwrite(gt_path, instances, compression="zlib")
163
164    return image_paths, gt_paths
165
166
167def get_cholec_instance_seg_dataset(
168    path: Union[str, os.PathLike],
169    patch_shape: Tuple[int, int],
170    split: Literal["train", "val", "test"],
171    resize_inputs: bool = False,
172    download: bool = False,
173    **kwargs
174) -> Dataset:
175    """Get the CholecInstanceSeg (Instance-CholecSeg8k subset) dataset for instrument instance segmentation.
176
177    Args:
178        path: Filepath to a folder where the data is downloaded for further processing.
179        patch_shape: The patch shape to use for training.
180        split: The choice of data split.
181        resize_inputs: Whether to resize inputs to the desired patch shape.
182        download: Whether to download the data if it is not present.
183        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
184
185    Returns:
186        The segmentation dataset.
187    """
188    image_paths, gt_paths = get_cholec_instance_seg_paths(path, split, download)
189
190    if resize_inputs:
191        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
192        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
193            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
194        )
195
196    return torch_em.default_segmentation_dataset(
197        raw_paths=image_paths,
198        raw_key=None,
199        label_paths=gt_paths,
200        label_key=None,
201        is_seg_dataset=False,
202        patch_shape=patch_shape,
203        **kwargs
204    )
205
206
207def get_cholec_instance_seg_loader(
208    path: Union[str, os.PathLike],
209    batch_size: int,
210    patch_shape: Tuple[int, int],
211    split: Literal["train", "val", "test"],
212    resize_inputs: bool = False,
213    download: bool = False,
214    **kwargs
215) -> DataLoader:
216    """Get the CholecInstanceSeg (Instance-CholecSeg8k subset) dataloader for instrument instance segmentation.
217
218    Args:
219        path: Filepath to a folder where the data is downloaded for further processing.
220        batch_size: The batch size for training.
221        patch_shape: The patch shape to use for training.
222        split: The choice of data split.
223        resize_inputs: Whether to resize inputs to the desired patch shape.
224        download: Whether to download the data if it is not present.
225        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
226
227    Returns:
228        The DataLoader.
229    """
230    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
231    dataset = get_cholec_instance_seg_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs)
232    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
ENTITY = 'syn66477541'
FRAME_PATTERN = re.compile('seg8k_video(\\d+)_(\\d+)\\.json')
def get_cholec_instance_seg_data(path: Union[os.PathLike, str], download: bool = False) -> Tuple[str, str]:
51def get_cholec_instance_seg_data(path: Union[os.PathLike, str], download: bool = False) -> Tuple[str, str]:
52    """Download the CholecInstanceSeg annotations (Instance-CholecSeg8k subset) and the raw CholecSeg8k frames.
53
54    Follow the instructions below to get access to the Synapse-hosted annotations.
55    - Create a free account at https://www.synapse.org.
56    - Generate a personal access token and store it in a '~/.synapseConfig' file, see
57      https://python-docs.synapse.org/tutorials/authentication/ for details.
58    - Install the 'synapseclient' python library.
59
60    Args:
61        path: Filepath to a folder where the data is downloaded for further processing.
62        download: Whether to download the data if it is not present.
63
64    Returns:
65        Filepath to the extracted CholecInstanceSeg annotations.
66        Filepath to the raw CholecSeg8k frames.
67    """
68    ann_dir = os.path.join(path, "cholecinstanceseg")
69    if not os.path.exists(ann_dir):
70        os.makedirs(path, exist_ok=True)
71
72        import synapseclient
73
74        syn = synapseclient.Synapse()
75        syn.login()
76        zip_path = os.path.join(path, "cholecinstanceseg.zip")
77        if not os.path.exists(zip_path):
78            if not download:
79                raise RuntimeError(f"Cannot find the data at {zip_path}, but download was set to False.")
80            syn.get(ENTITY, downloadLocation=path, downloadFile=True)
81
82        util.unzip(zip_path=zip_path, dst=path, remove=False)
83
84    raw_data_dir = get_cholecseg8k_data(path, download)
85
86    return ann_dir, raw_data_dir

Download the CholecInstanceSeg annotations (Instance-CholecSeg8k subset) and the raw CholecSeg8k frames.

Follow the instructions below to get access to the Synapse-hosted annotations.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath to the extracted CholecInstanceSeg annotations. Filepath to the raw CholecSeg8k frames.

def get_cholec_instance_seg_paths( path: Union[os.PathLike, str], split: Literal['train', 'val', 'test'], download: bool = False) -> Tuple[List[str], List[str]]:
 98def get_cholec_instance_seg_paths(
 99    path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False
100) -> Tuple[List[str], List[str]]:
101    """Get paths for the CholecInstanceSeg (Instance-CholecSeg8k subset) dataset.
102
103    Args:
104        path: Filepath to a folder where the data is downloaded for further processing.
105        split: The choice of data split.
106        download: Whether to download the data if it is not present.
107
108    Returns:
109        List of filepaths for the image data.
110        List of filepaths for the label data.
111    """
112    if split not in ["train", "val", "test"]:
113        raise ValueError(f"'{split}' is not a valid split. Please choose from 'train', 'val' or 'test'.")
114
115    ann_dir, raw_data_dir = get_cholec_instance_seg_data(path, download)
116
117    json_paths = natsorted(glob(os.path.join(ann_dir, split, "VID*_seg8k", "ann_dir", "*.json")))
118    assert len(json_paths) > 0, f"No annotations were found at '{os.path.join(ann_dir, split)}'."
119
120    ppdir = os.path.join(ann_dir, "preprocessed", split)
121    os.makedirs(os.path.join(ppdir, "images"), exist_ok=True)
122    os.makedirs(os.path.join(ppdir, "masks"), exist_ok=True)
123
124    image_paths, gt_paths = [], []
125    for json_path in tqdm(json_paths, desc=f"Preprocessing CholecInstanceSeg '{split}' split"):
126        match = FRAME_PATTERN.match(os.path.basename(json_path))
127        assert match is not None, f"Unexpected annotation filename: '{json_path}'."
128        video_id, frame_id = match.group(1), int(match.group(2))
129
130        org_image_paths = glob(
131            os.path.join(raw_data_dir, f"video{int(video_id):02d}", "video*", f"frame_{frame_id}_endo.png")
132        )
133        assert len(org_image_paths) == 1, (
134            f"Expected exactly one matching CholecSeg8k frame for video {video_id}, frame {frame_id}, "
135            f"found {len(org_image_paths)}."
136        )
137        org_image_path = org_image_paths[0]
138
139        image_id = os.path.basename(org_image_path)
140        image_path = os.path.join(ppdir, "images", image_id)
141        gt_path = os.path.join(ppdir, "masks", Path(image_id).with_suffix(".tif"))
142
143        image_paths.append(image_path)
144        gt_paths.append(gt_path)
145
146        if os.path.exists(image_path) and os.path.exists(gt_path):
147            continue
148
149        if not os.path.exists(image_path):
150            os.symlink(os.path.abspath(org_image_path), image_path)
151
152        with open(json_path) as f:
153            annotation = json.load(f)
154
155        # Frames without any annotated instrument instances lack the 'imageHeight' / 'imageWidth' keys,
156        # so the raw image shape is used as a fallback for the rasterized mask shape.
157        if "imageHeight" in annotation and "imageWidth" in annotation:
158            shape = (annotation["imageHeight"], annotation["imageWidth"])
159        else:
160            shape = imageio.imread(org_image_path).shape[:2]
161
162        instances = _rasterize_instances(annotation, shape)
163        imageio.imwrite(gt_path, instances, compression="zlib")
164
165    return image_paths, gt_paths

Get paths for the CholecInstanceSeg (Instance-CholecSeg8k subset) dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The choice of data split.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_cholec_instance_seg_dataset( path: Union[str, os.PathLike], patch_shape: Tuple[int, int], split: Literal['train', 'val', 'test'], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
168def get_cholec_instance_seg_dataset(
169    path: Union[str, os.PathLike],
170    patch_shape: Tuple[int, int],
171    split: Literal["train", "val", "test"],
172    resize_inputs: bool = False,
173    download: bool = False,
174    **kwargs
175) -> Dataset:
176    """Get the CholecInstanceSeg (Instance-CholecSeg8k subset) dataset for instrument instance segmentation.
177
178    Args:
179        path: Filepath to a folder where the data is downloaded for further processing.
180        patch_shape: The patch shape to use for training.
181        split: The choice of data split.
182        resize_inputs: Whether to resize inputs to the desired patch shape.
183        download: Whether to download the data if it is not present.
184        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
185
186    Returns:
187        The segmentation dataset.
188    """
189    image_paths, gt_paths = get_cholec_instance_seg_paths(path, split, download)
190
191    if resize_inputs:
192        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
193        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
194            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
195        )
196
197    return torch_em.default_segmentation_dataset(
198        raw_paths=image_paths,
199        raw_key=None,
200        label_paths=gt_paths,
201        label_key=None,
202        is_seg_dataset=False,
203        patch_shape=patch_shape,
204        **kwargs
205    )

Get the CholecInstanceSeg (Instance-CholecSeg8k subset) dataset for instrument instance segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_cholec_instance_seg_loader( path: Union[str, os.PathLike], batch_size: int, patch_shape: Tuple[int, int], split: Literal['train', 'val', 'test'], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
208def get_cholec_instance_seg_loader(
209    path: Union[str, os.PathLike],
210    batch_size: int,
211    patch_shape: Tuple[int, int],
212    split: Literal["train", "val", "test"],
213    resize_inputs: bool = False,
214    download: bool = False,
215    **kwargs
216) -> DataLoader:
217    """Get the CholecInstanceSeg (Instance-CholecSeg8k subset) dataloader for instrument instance segmentation.
218
219    Args:
220        path: Filepath to a folder where the data is downloaded for further processing.
221        batch_size: The batch size for training.
222        patch_shape: The patch shape to use for training.
223        split: The choice of data split.
224        resize_inputs: Whether to resize inputs to the desired patch shape.
225        download: Whether to download the data if it is not present.
226        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
227
228    Returns:
229        The DataLoader.
230    """
231    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
232    dataset = get_cholec_instance_seg_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs)
233    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the CholecInstanceSeg (Instance-CholecSeg8k subset) dataloader for instrument instance segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.