torch_em.data.datasets.medical.migs

The MIGS surgical navigation dataset contains annotations for ocular anatomy (corneal limbus, iris, trabecular meshwork) and surgical instrument segmentation in minimally invasive glaucoma surgery (MIGS) videos.

NOTE: The Zenodo record holds two dataset families:

  • 'MIGS video dataset .zip' (Task I): raw phase-recognition videos, unannotated.
  • 'Task II Raw Data.zip' / 'Task II Annotated Data.zip' (Task II): densely-annotated frames for semantic segmentation, which is what this module exposes.

Only a subset of the Task II frames (those from the training and validation splits of the original paper) have their grayscale class-index masks publicly released in 'Task II Annotated Data.zip': 4,062 annotated frames across 51 video clips from 39 patients, confirmed by inspecting the real archive contents, not assumed.

The dataset is located at https://doi.org/10.5281/zenodo.19438128 and is licensed under CC-BY-4.0.

This dataset is from the publication https://doi.org/10.1038/s41597-026-07535-2. Please cite it if you use this dataset for your research.

  1"""The MIGS surgical navigation dataset contains annotations for ocular anatomy
  2(corneal limbus, iris, trabecular meshwork) and surgical instrument segmentation
  3in minimally invasive glaucoma surgery (MIGS) videos.
  4
  5NOTE: The Zenodo record holds two dataset families:
  6- 'MIGS video dataset <i>.zip' (Task I): raw phase-recognition videos, unannotated.
  7- 'Task II Raw Data.zip' / 'Task II Annotated Data.zip' (Task II): densely-annotated
  8  frames for semantic segmentation, which is what this module exposes.
  9
 10Only a subset of the Task II frames (those from the training and validation splits
 11of the original paper) have their grayscale class-index masks publicly released in
 12'Task II Annotated Data.zip': 4,062 annotated frames across 51 video clips from 39
 13patients, confirmed by inspecting the real archive contents, not assumed.
 14
 15The dataset is located at https://doi.org/10.5281/zenodo.19438128 and is licensed
 16under CC-BY-4.0.
 17
 18This dataset is from the publication https://doi.org/10.1038/s41597-026-07535-2.
 19Please cite it if you use this dataset for your research.
 20"""
 21
 22import os
 23from glob import glob
 24from pathlib import Path
 25from natsort import natsorted
 26from typing import Union, Tuple, List
 27
 28from torch.utils.data import Dataset, DataLoader
 29
 30import torch_em
 31
 32from .. import util
 33
 34
 35URLS = {
 36    "raw": "https://zenodo.org/records/19438128/files/Task%20II%20Raw%20Data.zip",
 37    "annotations": "https://zenodo.org/records/19438128/files/Task%20II%20Annotated%20Data.zip",
 38}
 39
 40CHECKSUMS = {
 41    "raw": "7f45ee3f91edee52dc93a78dcb9e6b9d22796e6f7ae2f4a8eaa6c240271cde4d",
 42    "annotations": "94f6d4329c48104f1304eef85d10353c4b26efb232e3ac000db741d27d0c889b",
 43}
 44
 45
 46def get_migs_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 47    """Download the MIGS surgical navigation data.
 48
 49    Args:
 50        path: Filepath to a folder where the data is downloaded for further processing.
 51        download: Whether to download the data if it is not present.
 52
 53    Returns:
 54        Filepath where the data is downloaded.
 55    """
 56    raw_dir = os.path.join(path, "raw")
 57    annotations_dir = os.path.join(path, "annotations")
 58    if os.path.exists(raw_dir) and os.path.exists(annotations_dir):
 59        return path
 60
 61    os.makedirs(path, exist_ok=True)
 62
 63    raw_zip_path = os.path.join(path, "Task_II_Raw_Data.zip")
 64    util.download_source(path=raw_zip_path, url=URLS["raw"], download=download, checksum=CHECKSUMS["raw"])
 65    util.unzip(zip_path=raw_zip_path, dst=raw_dir)
 66
 67    annotations_zip_path = os.path.join(path, "Task_II_Annotated_Data.zip")
 68    util.download_source(
 69        path=annotations_zip_path, url=URLS["annotations"], download=download, checksum=CHECKSUMS["annotations"]
 70    )
 71    util.unzip(zip_path=annotations_zip_path, dst=annotations_dir)
 72
 73    return path
 74
 75
 76def get_migs_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 77    """Get paths to the MIGS surgical navigation data.
 78
 79    Args:
 80        path: Filepath to a folder where the data is downloaded for further processing.
 81        download: Whether to download the data if it is not present.
 82
 83    Returns:
 84        List of filepaths for the image data.
 85        List of filepaths for the label data.
 86    """
 87    data_dir = get_migs_data(path, download)
 88
 89    all_gt_paths = natsorted(glob(os.path.join(data_dir, "annotations", "Grayscale Images", "*", "*", "*.png")))
 90
 91    # A handful of masks in the release have no corresponding raw frame (confirmed by inspecting the real
 92    # archive contents, e.g. 'S frame 30/154_S_O/154_S_O_frame_00038' has a mask but the raw frame is missing
 93    # from 'Task II Raw Data.zip'). Such masks are skipped rather than raising an error.
 94    image_paths, gt_paths = [], []
 95    for gt_path in all_gt_paths:
 96        # e.g. '<data_dir>/annotations/Grayscale Images/F frame 50/144_F_O/144_F_O_frame_00040.png' pairs with
 97        # '<data_dir>/raw/F frame 50/144_F_O/144_F_O_frame_00040.jpg'.
 98        relpath = Path(gt_path).relative_to(os.path.join(data_dir, "annotations", "Grayscale Images"))
 99        image_path = os.path.join(data_dir, "raw", relpath.parent, f"{relpath.stem}.jpg")
100        if not os.path.exists(image_path):
101            continue
102
103        image_paths.append(image_path)
104        gt_paths.append(gt_path)
105
106    assert len(image_paths) == len(gt_paths) and len(image_paths) > 0, (
107        "No image-mask pairs were found. The expected 'raw/<category>/<video>/<frame>.jpg' vs "
108        "'annotations/Grayscale Images/<category>/<video>/<frame>.png' layout may not match the actual structure "
109        f"of the downloaded data. Please inspect the data at '{data_dir}'."
110    )
111
112    return image_paths, gt_paths
113
114
115def get_migs_dataset(
116    path: Union[os.PathLike, str],
117    patch_shape: Tuple[int, int],
118    resize_inputs: bool = False,
119    download: bool = False,
120    **kwargs
121) -> Dataset:
122    """Get the MIGS dataset for ocular anatomy and surgical instrument segmentation.
123
124    Args:
125        path: Filepath to a folder where the data is downloaded for further processing.
126        patch_shape: The patch shape to use for training.
127        resize_inputs: Whether to resize inputs to the desired patch shape.
128        download: Whether to download the data if it is not present.
129        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
130
131    Returns:
132        The segmentation dataset.
133    """
134    image_paths, gt_paths = get_migs_paths(path, download)
135
136    if resize_inputs:
137        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
138        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
139            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
140        )
141
142    return torch_em.default_segmentation_dataset(
143        raw_paths=image_paths,
144        raw_key=None,
145        label_paths=gt_paths,
146        label_key=None,
147        patch_shape=patch_shape,
148        is_seg_dataset=False,
149        **kwargs
150    )
151
152
153def get_migs_loader(
154    path: Union[os.PathLike, str],
155    batch_size: int,
156    patch_shape: Tuple[int, int],
157    resize_inputs: bool = False,
158    download: bool = False,
159    **kwargs
160) -> DataLoader:
161    """Get the MIGS dataloader for ocular anatomy and surgical instrument segmentation.
162
163    Args:
164        path: Filepath to a folder where the data is downloaded for further processing.
165        batch_size: The batch size for training.
166        patch_shape: The patch shape to use for training.
167        resize_inputs: Whether to resize inputs to the desired patch shape.
168        download: Whether to download the data if it is not present.
169        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
170
171    Returns:
172        The DataLoader.
173    """
174    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
175    dataset = get_migs_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
176    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URLS = {'raw': 'https://zenodo.org/records/19438128/files/Task%20II%20Raw%20Data.zip', 'annotations': 'https://zenodo.org/records/19438128/files/Task%20II%20Annotated%20Data.zip'}
CHECKSUMS = {'raw': '7f45ee3f91edee52dc93a78dcb9e6b9d22796e6f7ae2f4a8eaa6c240271cde4d', 'annotations': '94f6d4329c48104f1304eef85d10353c4b26efb232e3ac000db741d27d0c889b'}
def get_migs_data(path: Union[os.PathLike, str], download: bool = False) -> str:
47def get_migs_data(path: Union[os.PathLike, str], download: bool = False) -> str:
48    """Download the MIGS surgical navigation data.
49
50    Args:
51        path: Filepath to a folder where the data is downloaded for further processing.
52        download: Whether to download the data if it is not present.
53
54    Returns:
55        Filepath where the data is downloaded.
56    """
57    raw_dir = os.path.join(path, "raw")
58    annotations_dir = os.path.join(path, "annotations")
59    if os.path.exists(raw_dir) and os.path.exists(annotations_dir):
60        return path
61
62    os.makedirs(path, exist_ok=True)
63
64    raw_zip_path = os.path.join(path, "Task_II_Raw_Data.zip")
65    util.download_source(path=raw_zip_path, url=URLS["raw"], download=download, checksum=CHECKSUMS["raw"])
66    util.unzip(zip_path=raw_zip_path, dst=raw_dir)
67
68    annotations_zip_path = os.path.join(path, "Task_II_Annotated_Data.zip")
69    util.download_source(
70        path=annotations_zip_path, url=URLS["annotations"], download=download, checksum=CHECKSUMS["annotations"]
71    )
72    util.unzip(zip_path=annotations_zip_path, dst=annotations_dir)
73
74    return path

Download the MIGS surgical navigation data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_migs_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 77def get_migs_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 78    """Get paths to the MIGS surgical navigation data.
 79
 80    Args:
 81        path: Filepath to a folder where the data is downloaded for further processing.
 82        download: Whether to download the data if it is not present.
 83
 84    Returns:
 85        List of filepaths for the image data.
 86        List of filepaths for the label data.
 87    """
 88    data_dir = get_migs_data(path, download)
 89
 90    all_gt_paths = natsorted(glob(os.path.join(data_dir, "annotations", "Grayscale Images", "*", "*", "*.png")))
 91
 92    # A handful of masks in the release have no corresponding raw frame (confirmed by inspecting the real
 93    # archive contents, e.g. 'S frame 30/154_S_O/154_S_O_frame_00038' has a mask but the raw frame is missing
 94    # from 'Task II Raw Data.zip'). Such masks are skipped rather than raising an error.
 95    image_paths, gt_paths = [], []
 96    for gt_path in all_gt_paths:
 97        # e.g. '<data_dir>/annotations/Grayscale Images/F frame 50/144_F_O/144_F_O_frame_00040.png' pairs with
 98        # '<data_dir>/raw/F frame 50/144_F_O/144_F_O_frame_00040.jpg'.
 99        relpath = Path(gt_path).relative_to(os.path.join(data_dir, "annotations", "Grayscale Images"))
100        image_path = os.path.join(data_dir, "raw", relpath.parent, f"{relpath.stem}.jpg")
101        if not os.path.exists(image_path):
102            continue
103
104        image_paths.append(image_path)
105        gt_paths.append(gt_path)
106
107    assert len(image_paths) == len(gt_paths) and len(image_paths) > 0, (
108        "No image-mask pairs were found. The expected 'raw/<category>/<video>/<frame>.jpg' vs "
109        "'annotations/Grayscale Images/<category>/<video>/<frame>.png' layout may not match the actual structure "
110        f"of the downloaded data. Please inspect the data at '{data_dir}'."
111    )
112
113    return image_paths, gt_paths

Get paths to the MIGS surgical navigation data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_migs_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
116def get_migs_dataset(
117    path: Union[os.PathLike, str],
118    patch_shape: Tuple[int, int],
119    resize_inputs: bool = False,
120    download: bool = False,
121    **kwargs
122) -> Dataset:
123    """Get the MIGS dataset for ocular anatomy and surgical instrument segmentation.
124
125    Args:
126        path: Filepath to a folder where the data is downloaded for further processing.
127        patch_shape: The patch shape to use for training.
128        resize_inputs: Whether to resize inputs to the desired patch shape.
129        download: Whether to download the data if it is not present.
130        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
131
132    Returns:
133        The segmentation dataset.
134    """
135    image_paths, gt_paths = get_migs_paths(path, download)
136
137    if resize_inputs:
138        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
139        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
140            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
141        )
142
143    return torch_em.default_segmentation_dataset(
144        raw_paths=image_paths,
145        raw_key=None,
146        label_paths=gt_paths,
147        label_key=None,
148        patch_shape=patch_shape,
149        is_seg_dataset=False,
150        **kwargs
151    )

Get the MIGS dataset for ocular anatomy and surgical instrument segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_migs_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
154def get_migs_loader(
155    path: Union[os.PathLike, str],
156    batch_size: int,
157    patch_shape: Tuple[int, int],
158    resize_inputs: bool = False,
159    download: bool = False,
160    **kwargs
161) -> DataLoader:
162    """Get the MIGS dataloader for ocular anatomy and surgical instrument segmentation.
163
164    Args:
165        path: Filepath to a folder where the data is downloaded for further processing.
166        batch_size: The batch size for training.
167        patch_shape: The patch shape to use for training.
168        resize_inputs: Whether to resize inputs to the desired patch shape.
169        download: Whether to download the data if it is not present.
170        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
171
172    Returns:
173        The DataLoader.
174    """
175    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
176    dataset = get_migs_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
177    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the MIGS dataloader for ocular anatomy and surgical instrument segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize inputs to the desired patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.