torch_em.data.datasets.medical.refuge

The REFUGE dataset contains annotations for optic disc and optic cup segmentation in Fundus images, for the task of glaucoma assessment.

The dataset was published in the "Retinal Fundus Glaucoma Challenge" (REFUGE), organized as part of the 5th MICCAI Workshop on Ophthalmic Medical Image Analysis (OMIA) at MICCAI 2018. It comprises 1200 fundus images (400 for training, validation and test each), with manual pixel-wise annotations of the optic disc and optic cup, obtained by merging the annotations of seven independent glaucoma specialists from the Zhongshan Ophthalmic Center, Sun Yat-sen University, China.

The original data is hosted at https://refuge.grand-challenge.org, but this hosting has become stale. This dataloader uses a mirror of the data hosted on Kaggle: https://www.kaggle.com/datasets/victorlemosml/refuge2 (the 'REFUGE2' folder in this mirror corresponds to the original 2018 REFUGE data, not the REFUGE2 challenge).

The label masks are grayscale images (bmp for the train and test splits, png for the validation split) with 3 pixel values: 0 (optic cup), 128 (optic disc, excluding the cup) and 255 (background).

The dataset is from the publication https://doi.org/10.1016/j.media.2019.101570. Please cite it if you use this dataset for your research.

  1"""The REFUGE dataset contains annotations for optic disc and optic cup segmentation
  2in Fundus images, for the task of glaucoma assessment.
  3
  4The dataset was published in the "Retinal Fundus Glaucoma Challenge" (REFUGE), organized as part
  5of the 5th MICCAI Workshop on Ophthalmic Medical Image Analysis (OMIA) at MICCAI 2018.
  6It comprises 1200 fundus images (400 for training, validation and test each), with manual pixel-wise
  7annotations of the optic disc and optic cup, obtained by merging the annotations of seven independent
  8glaucoma specialists from the Zhongshan Ophthalmic Center, Sun Yat-sen University, China.
  9
 10The original data is hosted at https://refuge.grand-challenge.org, but this hosting has become stale.
 11This dataloader uses a mirror of the data hosted on Kaggle: https://www.kaggle.com/datasets/victorlemosml/refuge2
 12(the 'REFUGE2' folder in this mirror corresponds to the original 2018 REFUGE data, not the REFUGE2 challenge).
 13
 14The label masks are grayscale images (bmp for the train and test splits, png for the validation split) with
 153 pixel values: 0 (optic cup), 128 (optic disc, excluding the cup) and 255 (background).
 16
 17The dataset is from the publication https://doi.org/10.1016/j.media.2019.101570.
 18Please cite it if you use this dataset for your research.
 19"""
 20
 21import os
 22from glob import glob
 23from tqdm import tqdm
 24from pathlib import Path
 25from typing import Union, Tuple, Literal, List
 26
 27import imageio.v3 as imageio
 28
 29from torch.utils.data import Dataset, DataLoader
 30
 31import torch_em
 32
 33from .. import util
 34
 35
 36DATASET_NAME = "victorlemosml/refuge2"
 37
 38
 39def get_refuge_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 40    """Download the REFUGE dataset.
 41
 42    Args:
 43        path: Filepath to a folder where the data is downloaded for further processing.
 44        download: Whether to download the data if it is not present.
 45
 46    Returns:
 47        Filepath where the data is downloaded.
 48    """
 49    data_dir = os.path.join(path, "REFUGE2")
 50    if os.path.exists(data_dir):
 51        return data_dir
 52
 53    os.makedirs(path, exist_ok=True)
 54
 55    util.download_source_kaggle(path=path, dataset_name=DATASET_NAME, download=download)
 56    zip_path = os.path.join(path, "refuge2.zip")
 57    util.unzip(zip_path=zip_path, dst=path)
 58
 59    return data_dir
 60
 61
 62def _preprocess_labels(data_dir, mask_paths, task):
 63    gt_dir = os.path.join(data_dir, f"gt_{task}")
 64    os.makedirs(gt_dir, exist_ok=True)
 65
 66    gt_paths = []
 67    for mask_path in tqdm(mask_paths, desc=f"Preprocessing labels for '{task}'"):
 68        gt_path = os.path.join(gt_dir, f"{Path(mask_path).stem}.tif")
 69        gt_paths.append(gt_path)
 70        if os.path.exists(gt_path):
 71            continue
 72
 73        mask = imageio.imread(mask_path)
 74        if task == "disc":  # The optic disc region includes the optic cup.
 75            labels = (mask < 255).astype("uint8")
 76        else:  # The optic cup is the innermost region, marked with the pixel value 0.
 77            labels = (mask == 0).astype("uint8")
 78
 79        imageio.imwrite(gt_path, labels)
 80
 81    return gt_paths
 82
 83
 84def get_refuge_paths(
 85    path: Union[os.PathLike, str],
 86    split: Literal["train", "val", "test"],
 87    task: Literal["disc", "cup"] = "disc",
 88    download: bool = False,
 89) -> Tuple[List[str], List[str]]:
 90    """Get paths to the REFUGE data.
 91
 92    Args:
 93        path: Filepath to a folder where the data is downloaded for further processing.
 94        split: The choice of data split.
 95        task: The choice of labels for the specific task.
 96        download: Whether to download the data if it is not present.
 97
 98    Returns:
 99        List of filepaths for the image data.
100        List of filepaths for the label data.
101    """
102    data_dir = get_refuge_data(path=path, download=download)
103
104    assert split in ["train", "val", "test"], f"'{split}' is not a valid split."
105    assert task in ["disc", "cup"], f"'{task}' is not a valid task."
106
107    image_paths = sorted(glob(os.path.join(data_dir, split, "images", "*.jpg")))
108    mask_paths = sorted(
109        glob(os.path.join(data_dir, split, "mask", "*.bmp")) + glob(os.path.join(data_dir, split, "mask", "*.png"))
110    )
111    assert len(image_paths) == len(mask_paths) and len(image_paths) > 0
112
113    gt_paths = _preprocess_labels(data_dir, mask_paths, task)
114
115    return image_paths, gt_paths
116
117
118def get_refuge_dataset(
119    path: Union[os.PathLike, str],
120    patch_shape: Tuple[int, int],
121    split: Literal["train", "val", "test"],
122    task: Literal["disc", "cup"] = "disc",
123    resize_inputs: bool = False,
124    download: bool = False,
125    **kwargs
126) -> Dataset:
127    """Get the REFUGE dataset for segmentation of optic disc and optic cup in fundus images.
128
129    Args:
130        path: Filepath to a folder where the data is downloaded for further processing.
131        patch_shape: The patch shape to use for training.
132        split: The choice of data split.
133        task: The choice of labels for the specific task.
134        resize_inputs: Whether to resize the inputs to the expected patch shape.
135        download: Whether to download the data if it is not present.
136        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
137
138    Returns:
139        The segmentation dataset.
140    """
141    image_paths, gt_paths = get_refuge_paths(path, split, task, download)
142
143    if resize_inputs:
144        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
145        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
146            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
147        )
148
149    return torch_em.default_segmentation_dataset(
150        raw_paths=image_paths,
151        raw_key=None,
152        label_paths=gt_paths,
153        label_key=None,
154        patch_shape=patch_shape,
155        is_seg_dataset=False,
156        **kwargs
157    )
158
159
160def get_refuge_loader(
161    path: Union[os.PathLike, str],
162    batch_size: int,
163    patch_shape: Tuple[int, int],
164    split: Literal["train", "val", "test"],
165    task: Literal["disc", "cup"] = "disc",
166    resize_inputs: bool = False,
167    download: bool = False,
168    **kwargs
169) -> DataLoader:
170    """Get the REFUGE dataloader for segmentation of optic disc and optic cup in fundus images.
171
172    Args:
173        path: Filepath to a folder where the data is downloaded for further processing.
174        batch_size: The batch size for training.
175        patch_shape: The patch shape to use for training.
176        split: The choice of data split.
177        task: The choice of labels for the specific task.
178        resize_inputs: Whether to resize the inputs to the expected patch shape.
179        download: Whether to download the data if it is not present.
180        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
181
182    Returns:
183        The DataLoader.
184    """
185    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
186    dataset = get_refuge_dataset(path, patch_shape, split, task, resize_inputs, download, **ds_kwargs)
187    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
DATASET_NAME = 'victorlemosml/refuge2'
def get_refuge_data(path: Union[os.PathLike, str], download: bool = False) -> str:
40def get_refuge_data(path: Union[os.PathLike, str], download: bool = False) -> str:
41    """Download the REFUGE dataset.
42
43    Args:
44        path: Filepath to a folder where the data is downloaded for further processing.
45        download: Whether to download the data if it is not present.
46
47    Returns:
48        Filepath where the data is downloaded.
49    """
50    data_dir = os.path.join(path, "REFUGE2")
51    if os.path.exists(data_dir):
52        return data_dir
53
54    os.makedirs(path, exist_ok=True)
55
56    util.download_source_kaggle(path=path, dataset_name=DATASET_NAME, download=download)
57    zip_path = os.path.join(path, "refuge2.zip")
58    util.unzip(zip_path=zip_path, dst=path)
59
60    return data_dir

Download the REFUGE dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_refuge_paths( path: Union[os.PathLike, str], split: Literal['train', 'val', 'test'], task: Literal['disc', 'cup'] = 'disc', download: bool = False) -> Tuple[List[str], List[str]]:
 85def get_refuge_paths(
 86    path: Union[os.PathLike, str],
 87    split: Literal["train", "val", "test"],
 88    task: Literal["disc", "cup"] = "disc",
 89    download: bool = False,
 90) -> Tuple[List[str], List[str]]:
 91    """Get paths to the REFUGE data.
 92
 93    Args:
 94        path: Filepath to a folder where the data is downloaded for further processing.
 95        split: The choice of data split.
 96        task: The choice of labels for the specific task.
 97        download: Whether to download the data if it is not present.
 98
 99    Returns:
100        List of filepaths for the image data.
101        List of filepaths for the label data.
102    """
103    data_dir = get_refuge_data(path=path, download=download)
104
105    assert split in ["train", "val", "test"], f"'{split}' is not a valid split."
106    assert task in ["disc", "cup"], f"'{task}' is not a valid task."
107
108    image_paths = sorted(glob(os.path.join(data_dir, split, "images", "*.jpg")))
109    mask_paths = sorted(
110        glob(os.path.join(data_dir, split, "mask", "*.bmp")) + glob(os.path.join(data_dir, split, "mask", "*.png"))
111    )
112    assert len(image_paths) == len(mask_paths) and len(image_paths) > 0
113
114    gt_paths = _preprocess_labels(data_dir, mask_paths, task)
115
116    return image_paths, gt_paths

Get paths to the REFUGE data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The choice of data split.
  • task: The choice of labels for the specific task.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_refuge_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], split: Literal['train', 'val', 'test'], task: Literal['disc', 'cup'] = 'disc', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
119def get_refuge_dataset(
120    path: Union[os.PathLike, str],
121    patch_shape: Tuple[int, int],
122    split: Literal["train", "val", "test"],
123    task: Literal["disc", "cup"] = "disc",
124    resize_inputs: bool = False,
125    download: bool = False,
126    **kwargs
127) -> Dataset:
128    """Get the REFUGE dataset for segmentation of optic disc and optic cup in fundus images.
129
130    Args:
131        path: Filepath to a folder where the data is downloaded for further processing.
132        patch_shape: The patch shape to use for training.
133        split: The choice of data split.
134        task: The choice of labels for the specific task.
135        resize_inputs: Whether to resize the inputs to the expected patch shape.
136        download: Whether to download the data if it is not present.
137        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
138
139    Returns:
140        The segmentation dataset.
141    """
142    image_paths, gt_paths = get_refuge_paths(path, split, task, download)
143
144    if resize_inputs:
145        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
146        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
147            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
148        )
149
150    return torch_em.default_segmentation_dataset(
151        raw_paths=image_paths,
152        raw_key=None,
153        label_paths=gt_paths,
154        label_key=None,
155        patch_shape=patch_shape,
156        is_seg_dataset=False,
157        **kwargs
158    )

Get the REFUGE dataset for segmentation of optic disc and optic cup in fundus images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split.
  • task: The choice of labels for the specific task.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_refuge_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], split: Literal['train', 'val', 'test'], task: Literal['disc', 'cup'] = 'disc', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
161def get_refuge_loader(
162    path: Union[os.PathLike, str],
163    batch_size: int,
164    patch_shape: Tuple[int, int],
165    split: Literal["train", "val", "test"],
166    task: Literal["disc", "cup"] = "disc",
167    resize_inputs: bool = False,
168    download: bool = False,
169    **kwargs
170) -> DataLoader:
171    """Get the REFUGE dataloader for segmentation of optic disc and optic cup in fundus images.
172
173    Args:
174        path: Filepath to a folder where the data is downloaded for further processing.
175        batch_size: The batch size for training.
176        patch_shape: The patch shape to use for training.
177        split: The choice of data split.
178        task: The choice of labels for the specific task.
179        resize_inputs: Whether to resize the inputs to the expected patch shape.
180        download: Whether to download the data if it is not present.
181        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
182
183    Returns:
184        The DataLoader.
185    """
186    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
187    dataset = get_refuge_dataset(path, patch_shape, split, task, resize_inputs, download, **ds_kwargs)
188    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the REFUGE dataloader for segmentation of optic disc and optic cup in fundus images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split.
  • task: The choice of labels for the specific task.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.