torch_em.data.datasets.medical.migs
The MIGS surgical navigation dataset contains annotations for ocular anatomy (corneal limbus, iris, trabecular meshwork) and surgical instrument segmentation in minimally invasive glaucoma surgery (MIGS) videos.
NOTE: The Zenodo record holds two dataset families:
- 'MIGS video dataset .zip' (Task I): raw phase-recognition videos, unannotated.
- 'Task II Raw Data.zip' / 'Task II Annotated Data.zip' (Task II): densely-annotated frames for semantic segmentation, which is what this module exposes.
Only a subset of the Task II frames (those from the training and validation splits of the original paper) have their grayscale class-index masks publicly released in 'Task II Annotated Data.zip': 4,062 annotated frames across 51 video clips from 39 patients, confirmed by inspecting the real archive contents, not assumed.
The dataset is located at https://doi.org/10.5281/zenodo.19438128 and is licensed under CC-BY-4.0.
This dataset is from the publication https://doi.org/10.1038/s41597-026-07535-2. Please cite it if you use this dataset for your research.
1"""The MIGS surgical navigation dataset contains annotations for ocular anatomy 2(corneal limbus, iris, trabecular meshwork) and surgical instrument segmentation 3in minimally invasive glaucoma surgery (MIGS) videos. 4 5NOTE: The Zenodo record holds two dataset families: 6- 'MIGS video dataset <i>.zip' (Task I): raw phase-recognition videos, unannotated. 7- 'Task II Raw Data.zip' / 'Task II Annotated Data.zip' (Task II): densely-annotated 8 frames for semantic segmentation, which is what this module exposes. 9 10Only a subset of the Task II frames (those from the training and validation splits 11of the original paper) have their grayscale class-index masks publicly released in 12'Task II Annotated Data.zip': 4,062 annotated frames across 51 video clips from 39 13patients, confirmed by inspecting the real archive contents, not assumed. 14 15The dataset is located at https://doi.org/10.5281/zenodo.19438128 and is licensed 16under CC-BY-4.0. 17 18This dataset is from the publication https://doi.org/10.1038/s41597-026-07535-2. 19Please cite it if you use this dataset for your research. 20""" 21 22import os 23from glob import glob 24from pathlib import Path 25from natsort import natsorted 26from typing import Union, Tuple, List 27 28from torch.utils.data import Dataset, DataLoader 29 30import torch_em 31 32from .. import util 33 34 35URLS = { 36 "raw": "https://zenodo.org/records/19438128/files/Task%20II%20Raw%20Data.zip", 37 "annotations": "https://zenodo.org/records/19438128/files/Task%20II%20Annotated%20Data.zip", 38} 39 40CHECKSUMS = { 41 "raw": "7f45ee3f91edee52dc93a78dcb9e6b9d22796e6f7ae2f4a8eaa6c240271cde4d", 42 "annotations": "94f6d4329c48104f1304eef85d10353c4b26efb232e3ac000db741d27d0c889b", 43} 44 45 46def get_migs_data(path: Union[os.PathLike, str], download: bool = False) -> str: 47 """Download the MIGS surgical navigation data. 48 49 Args: 50 path: Filepath to a folder where the data is downloaded for further processing. 51 download: Whether to download the data if it is not present. 52 53 Returns: 54 Filepath where the data is downloaded. 55 """ 56 raw_dir = os.path.join(path, "raw") 57 annotations_dir = os.path.join(path, "annotations") 58 if os.path.exists(raw_dir) and os.path.exists(annotations_dir): 59 return path 60 61 os.makedirs(path, exist_ok=True) 62 63 raw_zip_path = os.path.join(path, "Task_II_Raw_Data.zip") 64 util.download_source(path=raw_zip_path, url=URLS["raw"], download=download, checksum=CHECKSUMS["raw"]) 65 util.unzip(zip_path=raw_zip_path, dst=raw_dir) 66 67 annotations_zip_path = os.path.join(path, "Task_II_Annotated_Data.zip") 68 util.download_source( 69 path=annotations_zip_path, url=URLS["annotations"], download=download, checksum=CHECKSUMS["annotations"] 70 ) 71 util.unzip(zip_path=annotations_zip_path, dst=annotations_dir) 72 73 return path 74 75 76def get_migs_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 77 """Get paths to the MIGS surgical navigation data. 78 79 Args: 80 path: Filepath to a folder where the data is downloaded for further processing. 81 download: Whether to download the data if it is not present. 82 83 Returns: 84 List of filepaths for the image data. 85 List of filepaths for the label data. 86 """ 87 data_dir = get_migs_data(path, download) 88 89 all_gt_paths = natsorted(glob(os.path.join(data_dir, "annotations", "Grayscale Images", "*", "*", "*.png"))) 90 91 # A handful of masks in the release have no corresponding raw frame (confirmed by inspecting the real 92 # archive contents, e.g. 'S frame 30/154_S_O/154_S_O_frame_00038' has a mask but the raw frame is missing 93 # from 'Task II Raw Data.zip'). Such masks are skipped rather than raising an error. 94 image_paths, gt_paths = [], [] 95 for gt_path in all_gt_paths: 96 # e.g. '<data_dir>/annotations/Grayscale Images/F frame 50/144_F_O/144_F_O_frame_00040.png' pairs with 97 # '<data_dir>/raw/F frame 50/144_F_O/144_F_O_frame_00040.jpg'. 98 relpath = Path(gt_path).relative_to(os.path.join(data_dir, "annotations", "Grayscale Images")) 99 image_path = os.path.join(data_dir, "raw", relpath.parent, f"{relpath.stem}.jpg") 100 if not os.path.exists(image_path): 101 continue 102 103 image_paths.append(image_path) 104 gt_paths.append(gt_path) 105 106 assert len(image_paths) == len(gt_paths) and len(image_paths) > 0, ( 107 "No image-mask pairs were found. The expected 'raw/<category>/<video>/<frame>.jpg' vs " 108 "'annotations/Grayscale Images/<category>/<video>/<frame>.png' layout may not match the actual structure " 109 f"of the downloaded data. Please inspect the data at '{data_dir}'." 110 ) 111 112 return image_paths, gt_paths 113 114 115def get_migs_dataset( 116 path: Union[os.PathLike, str], 117 patch_shape: Tuple[int, int], 118 resize_inputs: bool = False, 119 download: bool = False, 120 **kwargs 121) -> Dataset: 122 """Get the MIGS dataset for ocular anatomy and surgical instrument segmentation. 123 124 Args: 125 path: Filepath to a folder where the data is downloaded for further processing. 126 patch_shape: The patch shape to use for training. 127 resize_inputs: Whether to resize inputs to the desired patch shape. 128 download: Whether to download the data if it is not present. 129 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 130 131 Returns: 132 The segmentation dataset. 133 """ 134 image_paths, gt_paths = get_migs_paths(path, download) 135 136 if resize_inputs: 137 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 138 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 139 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 140 ) 141 142 return torch_em.default_segmentation_dataset( 143 raw_paths=image_paths, 144 raw_key=None, 145 label_paths=gt_paths, 146 label_key=None, 147 patch_shape=patch_shape, 148 is_seg_dataset=False, 149 **kwargs 150 ) 151 152 153def get_migs_loader( 154 path: Union[os.PathLike, str], 155 batch_size: int, 156 patch_shape: Tuple[int, int], 157 resize_inputs: bool = False, 158 download: bool = False, 159 **kwargs 160) -> DataLoader: 161 """Get the MIGS dataloader for ocular anatomy and surgical instrument segmentation. 162 163 Args: 164 path: Filepath to a folder where the data is downloaded for further processing. 165 batch_size: The batch size for training. 166 patch_shape: The patch shape to use for training. 167 resize_inputs: Whether to resize inputs to the desired patch shape. 168 download: Whether to download the data if it is not present. 169 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 170 171 Returns: 172 The DataLoader. 173 """ 174 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 175 dataset = get_migs_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 176 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
47def get_migs_data(path: Union[os.PathLike, str], download: bool = False) -> str: 48 """Download the MIGS surgical navigation data. 49 50 Args: 51 path: Filepath to a folder where the data is downloaded for further processing. 52 download: Whether to download the data if it is not present. 53 54 Returns: 55 Filepath where the data is downloaded. 56 """ 57 raw_dir = os.path.join(path, "raw") 58 annotations_dir = os.path.join(path, "annotations") 59 if os.path.exists(raw_dir) and os.path.exists(annotations_dir): 60 return path 61 62 os.makedirs(path, exist_ok=True) 63 64 raw_zip_path = os.path.join(path, "Task_II_Raw_Data.zip") 65 util.download_source(path=raw_zip_path, url=URLS["raw"], download=download, checksum=CHECKSUMS["raw"]) 66 util.unzip(zip_path=raw_zip_path, dst=raw_dir) 67 68 annotations_zip_path = os.path.join(path, "Task_II_Annotated_Data.zip") 69 util.download_source( 70 path=annotations_zip_path, url=URLS["annotations"], download=download, checksum=CHECKSUMS["annotations"] 71 ) 72 util.unzip(zip_path=annotations_zip_path, dst=annotations_dir) 73 74 return path
Download the MIGS surgical navigation data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
77def get_migs_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 78 """Get paths to the MIGS surgical navigation data. 79 80 Args: 81 path: Filepath to a folder where the data is downloaded for further processing. 82 download: Whether to download the data if it is not present. 83 84 Returns: 85 List of filepaths for the image data. 86 List of filepaths for the label data. 87 """ 88 data_dir = get_migs_data(path, download) 89 90 all_gt_paths = natsorted(glob(os.path.join(data_dir, "annotations", "Grayscale Images", "*", "*", "*.png"))) 91 92 # A handful of masks in the release have no corresponding raw frame (confirmed by inspecting the real 93 # archive contents, e.g. 'S frame 30/154_S_O/154_S_O_frame_00038' has a mask but the raw frame is missing 94 # from 'Task II Raw Data.zip'). Such masks are skipped rather than raising an error. 95 image_paths, gt_paths = [], [] 96 for gt_path in all_gt_paths: 97 # e.g. '<data_dir>/annotations/Grayscale Images/F frame 50/144_F_O/144_F_O_frame_00040.png' pairs with 98 # '<data_dir>/raw/F frame 50/144_F_O/144_F_O_frame_00040.jpg'. 99 relpath = Path(gt_path).relative_to(os.path.join(data_dir, "annotations", "Grayscale Images")) 100 image_path = os.path.join(data_dir, "raw", relpath.parent, f"{relpath.stem}.jpg") 101 if not os.path.exists(image_path): 102 continue 103 104 image_paths.append(image_path) 105 gt_paths.append(gt_path) 106 107 assert len(image_paths) == len(gt_paths) and len(image_paths) > 0, ( 108 "No image-mask pairs were found. The expected 'raw/<category>/<video>/<frame>.jpg' vs " 109 "'annotations/Grayscale Images/<category>/<video>/<frame>.png' layout may not match the actual structure " 110 f"of the downloaded data. Please inspect the data at '{data_dir}'." 111 ) 112 113 return image_paths, gt_paths
Get paths to the MIGS surgical navigation data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
116def get_migs_dataset( 117 path: Union[os.PathLike, str], 118 patch_shape: Tuple[int, int], 119 resize_inputs: bool = False, 120 download: bool = False, 121 **kwargs 122) -> Dataset: 123 """Get the MIGS dataset for ocular anatomy and surgical instrument segmentation. 124 125 Args: 126 path: Filepath to a folder where the data is downloaded for further processing. 127 patch_shape: The patch shape to use for training. 128 resize_inputs: Whether to resize inputs to the desired patch shape. 129 download: Whether to download the data if it is not present. 130 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 131 132 Returns: 133 The segmentation dataset. 134 """ 135 image_paths, gt_paths = get_migs_paths(path, download) 136 137 if resize_inputs: 138 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 139 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 140 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 141 ) 142 143 return torch_em.default_segmentation_dataset( 144 raw_paths=image_paths, 145 raw_key=None, 146 label_paths=gt_paths, 147 label_key=None, 148 patch_shape=patch_shape, 149 is_seg_dataset=False, 150 **kwargs 151 )
Get the MIGS dataset for ocular anatomy and surgical instrument segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
154def get_migs_loader( 155 path: Union[os.PathLike, str], 156 batch_size: int, 157 patch_shape: Tuple[int, int], 158 resize_inputs: bool = False, 159 download: bool = False, 160 **kwargs 161) -> DataLoader: 162 """Get the MIGS dataloader for ocular anatomy and surgical instrument segmentation. 163 164 Args: 165 path: Filepath to a folder where the data is downloaded for further processing. 166 batch_size: The batch size for training. 167 patch_shape: The patch shape to use for training. 168 resize_inputs: Whether to resize inputs to the desired patch shape. 169 download: Whether to download the data if it is not present. 170 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 171 172 Returns: 173 The DataLoader. 174 """ 175 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 176 dataset = get_migs_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 177 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the MIGS dataloader for ocular anatomy and surgical instrument segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize inputs to the desired patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.