torch_em.data.datasets.medical.hvdropdb

The HVDROPDB dataset contains annotations for optic disc, blood vessel, and demarcation line / ridge segmentation in fundus images of premature infants, for the task of retinopathy of prematurity (ROP) screening.

This dataset is located at https://doi.org/10.17632/xw5xc7xrmp.3, under the CC BY 4.0 license. The dataset is from the publication https://doi.org/10.1016/j.dib.2023.109839. Please cite it if you use this dataset for your research.

The images were acquired with two imaging systems (RetCam and Neo) at PBMA's H.V. Desai Eye Hospital, Pune, screening preterm infants for ROP. Ground truth masks were prepared manually (with Adobe Photoshop) by a group of ROP experts, separately for the optic disc, blood vessels, and the demarcation line / ridge.

  1"""The HVDROPDB dataset contains annotations for optic disc, blood vessel, and
  2demarcation line / ridge segmentation in fundus images of premature infants,
  3for the task of retinopathy of prematurity (ROP) screening.
  4
  5This dataset is located at https://doi.org/10.17632/xw5xc7xrmp.3, under the CC BY 4.0 license.
  6The dataset is from the publication https://doi.org/10.1016/j.dib.2023.109839.
  7Please cite it if you use this dataset for your research.
  8
  9The images were acquired with two imaging systems (RetCam and Neo) at PBMA's H.V. Desai Eye
 10Hospital, Pune, screening preterm infants for ROP. Ground truth masks were prepared manually
 11(with Adobe Photoshop) by a group of ROP experts, separately for the optic disc, blood vessels,
 12and the demarcation line / ridge.
 13"""
 14
 15import os
 16from glob import glob
 17from pathlib import Path
 18from natsort import natsorted
 19from typing import Union, Tuple, Literal, List
 20
 21import imageio.v3 as imageio
 22
 23from torch.utils.data import Dataset, DataLoader
 24
 25import torch_em
 26
 27from .. import util
 28
 29
 30URL = "https://data.mendeley.com/public-files/datasets/xw5xc7xrmp/files/dfa3ec48-c763-439f-905a-bb0904ea840a/file_downloaded"  # noqa
 31CHECKSUM = "0a3aa96bbe489fc2f9103bac80fdb3eddc4fc91533e87555446970e9d021c1d2"
 32
 33# Maps the structure to segment to the folder name and the image / mask subfolder prefixes
 34# used per imaging device inside the archive.
 35STRUCTURES = {
 36    "optic_disc": ("HVDROPDB-OD", {"RetCam": "Retcam_OpticDisc", "Neo": "Neo_OpticDisc"}),
 37    "vessels": ("HVDROPDB-BV", {"RetCam": "RetCam_Vessels", "Neo": "Neo_Vessels"}),
 38    "ridge": ("HVDROPDB-RIDGE", {"RetCam": "RetCam_Ridge", "Neo": "Neo_Ridge"}),
 39}
 40
 41
 42def get_hvdropdb_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 43    """Download the HVDROPDB dataset.
 44
 45    Args:
 46        path: Filepath to a folder where the data is downloaded for further processing.
 47        download: Whether to download the data if it is not present.
 48
 49    Returns:
 50        Filepath where the data is downloaded.
 51    """
 52    data_dir = os.path.join(path, "HVDROPDB_RetCam_Neo_Segmentation")
 53    if os.path.exists(data_dir):
 54        return data_dir
 55
 56    os.makedirs(path, exist_ok=True)
 57
 58    zip_path = os.path.join(path, "HVDROPDB_RetCam_Neo_Segmentation.zip")
 59    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 60    util.unzip(zip_path=zip_path, dst=path)
 61
 62    return data_dir
 63
 64
 65def _binarize_mask(mask_path, dst_dir):
 66    dst_path = os.path.join(dst_dir, Path(mask_path).stem + ".png")
 67    if os.path.exists(dst_path):
 68        return dst_path
 69
 70    os.makedirs(dst_dir, exist_ok=True)
 71    mask = imageio.imread(mask_path)
 72    if mask.ndim == 3:
 73        mask = mask[..., 0]
 74    mask = (mask > 127).astype("uint8")
 75    imageio.imwrite(dst_path, mask)
 76    return dst_path
 77
 78
 79def get_hvdropdb_paths(
 80    path: Union[os.PathLike, str],
 81    structure: Literal["optic_disc", "vessels", "ridge"],
 82    device: Literal["RetCam", "Neo"] = "RetCam",
 83    download: bool = False,
 84) -> Tuple[List[str], List[str]]:
 85    """Get paths to the HVDROPDB data.
 86
 87    Args:
 88        path: Filepath to a folder where the data is downloaded for further processing.
 89        structure: The choice of anatomical structure to segment.
 90        device: The choice of imaging device used to acquire the fundus images.
 91        download: Whether to download the data if it is not present.
 92
 93    Returns:
 94        List of filepaths for the image data.
 95        List of filepaths for the label data.
 96    """
 97    assert structure in STRUCTURES, f"'{structure}' is not a valid structure choice."
 98    assert device in ["RetCam", "Neo"], f"'{device}' is not a valid device choice."
 99
100    data_dir = get_hvdropdb_data(path=path, download=download)
101
102    folder, prefixes = STRUCTURES[structure]
103    prefix = prefixes[device]
104
105    image_dir = os.path.join(data_dir, folder, f"{prefix}_images")
106    gt_dir = os.path.join(data_dir, folder, f"{prefix}_masks")
107
108    image_paths = natsorted(glob(os.path.join(image_dir, "*.png")))
109    raw_gt_paths = natsorted(glob(os.path.join(gt_dir, "*.png")))
110
111    assert len(image_paths) == len(raw_gt_paths) and len(image_paths) > 0
112
113    # The shipped masks are RGB images (with an identical value per channel). We binarize
114    # them into single-channel masks here, as expected by `torch_em.default_segmentation_dataset`.
115    binary_gt_dir = os.path.join(data_dir, folder, f"{prefix}_masks_binary")
116    gt_paths = [_binarize_mask(p, binary_gt_dir) for p in raw_gt_paths]
117
118    return image_paths, gt_paths
119
120
121def get_hvdropdb_dataset(
122    path: Union[os.PathLike, str],
123    patch_shape: Tuple[int, int],
124    structure: Literal["optic_disc", "vessels", "ridge"],
125    device: Literal["RetCam", "Neo"] = "RetCam",
126    resize_inputs: bool = False,
127    download: bool = False,
128    **kwargs
129) -> Dataset:
130    """Get the HVDROPDB dataset for optic disc / vessel / demarcation line segmentation in fundus images.
131
132    Args:
133        path: Filepath to a folder where the downloaded data will be saved.
134        patch_shape: The patch shape to use for training.
135        structure: The choice of anatomical structure to segment.
136        device: The choice of imaging device used to acquire the fundus images.
137        resize_inputs: Whether to resize the inputs to the expected patch shape.
138        download: Whether to download the data if it is not present.
139        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
140
141    Returns:
142        The segmentation dataset.
143    """
144    image_paths, gt_paths = get_hvdropdb_paths(path=path, structure=structure, device=device, download=download)
145
146    if resize_inputs:
147        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
148        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
149            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs,
150        )
151
152    return torch_em.default_segmentation_dataset(
153        raw_paths=image_paths,
154        raw_key=None,
155        label_paths=gt_paths,
156        label_key=None,
157        patch_shape=patch_shape,
158        is_seg_dataset=False,
159        **kwargs
160    )
161
162
163def get_hvdropdb_loader(
164    path: Union[os.PathLike, str],
165    batch_size: int,
166    patch_shape: Tuple[int, int],
167    structure: Literal["optic_disc", "vessels", "ridge"],
168    device: Literal["RetCam", "Neo"] = "RetCam",
169    resize_inputs: bool = False,
170    download: bool = False,
171    **kwargs
172) -> DataLoader:
173    """Get the HVDROPDB dataloader for optic disc / vessel / demarcation line segmentation in fundus images.
174
175    Args:
176        path: Filepath to a folder where the downloaded data will be saved.
177        batch_size: The batch size for training.
178        patch_shape: The patch shape to use for training.
179        structure: The choice of anatomical structure to segment.
180        device: The choice of imaging device used to acquire the fundus images.
181        resize_inputs: Whether to resize the inputs to the expected patch shape.
182        download: Whether to download the data if it is not present.
183        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
184
185    Returns:
186        The DataLoader.
187    """
188    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
189    dataset = get_hvdropdb_dataset(path, patch_shape, structure, device, resize_inputs, download, **ds_kwargs)
190    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://data.mendeley.com/public-files/datasets/xw5xc7xrmp/files/dfa3ec48-c763-439f-905a-bb0904ea840a/file_downloaded'
CHECKSUM = '0a3aa96bbe489fc2f9103bac80fdb3eddc4fc91533e87555446970e9d021c1d2'
STRUCTURES = {'optic_disc': ('HVDROPDB-OD', {'RetCam': 'Retcam_OpticDisc', 'Neo': 'Neo_OpticDisc'}), 'vessels': ('HVDROPDB-BV', {'RetCam': 'RetCam_Vessels', 'Neo': 'Neo_Vessels'}), 'ridge': ('HVDROPDB-RIDGE', {'RetCam': 'RetCam_Ridge', 'Neo': 'Neo_Ridge'})}
def get_hvdropdb_data(path: Union[os.PathLike, str], download: bool = False) -> str:
43def get_hvdropdb_data(path: Union[os.PathLike, str], download: bool = False) -> str:
44    """Download the HVDROPDB dataset.
45
46    Args:
47        path: Filepath to a folder where the data is downloaded for further processing.
48        download: Whether to download the data if it is not present.
49
50    Returns:
51        Filepath where the data is downloaded.
52    """
53    data_dir = os.path.join(path, "HVDROPDB_RetCam_Neo_Segmentation")
54    if os.path.exists(data_dir):
55        return data_dir
56
57    os.makedirs(path, exist_ok=True)
58
59    zip_path = os.path.join(path, "HVDROPDB_RetCam_Neo_Segmentation.zip")
60    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
61    util.unzip(zip_path=zip_path, dst=path)
62
63    return data_dir

Download the HVDROPDB dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_hvdropdb_paths( path: Union[os.PathLike, str], structure: Literal['optic_disc', 'vessels', 'ridge'], device: Literal['RetCam', 'Neo'] = 'RetCam', download: bool = False) -> Tuple[List[str], List[str]]:
 80def get_hvdropdb_paths(
 81    path: Union[os.PathLike, str],
 82    structure: Literal["optic_disc", "vessels", "ridge"],
 83    device: Literal["RetCam", "Neo"] = "RetCam",
 84    download: bool = False,
 85) -> Tuple[List[str], List[str]]:
 86    """Get paths to the HVDROPDB data.
 87
 88    Args:
 89        path: Filepath to a folder where the data is downloaded for further processing.
 90        structure: The choice of anatomical structure to segment.
 91        device: The choice of imaging device used to acquire the fundus images.
 92        download: Whether to download the data if it is not present.
 93
 94    Returns:
 95        List of filepaths for the image data.
 96        List of filepaths for the label data.
 97    """
 98    assert structure in STRUCTURES, f"'{structure}' is not a valid structure choice."
 99    assert device in ["RetCam", "Neo"], f"'{device}' is not a valid device choice."
100
101    data_dir = get_hvdropdb_data(path=path, download=download)
102
103    folder, prefixes = STRUCTURES[structure]
104    prefix = prefixes[device]
105
106    image_dir = os.path.join(data_dir, folder, f"{prefix}_images")
107    gt_dir = os.path.join(data_dir, folder, f"{prefix}_masks")
108
109    image_paths = natsorted(glob(os.path.join(image_dir, "*.png")))
110    raw_gt_paths = natsorted(glob(os.path.join(gt_dir, "*.png")))
111
112    assert len(image_paths) == len(raw_gt_paths) and len(image_paths) > 0
113
114    # The shipped masks are RGB images (with an identical value per channel). We binarize
115    # them into single-channel masks here, as expected by `torch_em.default_segmentation_dataset`.
116    binary_gt_dir = os.path.join(data_dir, folder, f"{prefix}_masks_binary")
117    gt_paths = [_binarize_mask(p, binary_gt_dir) for p in raw_gt_paths]
118
119    return image_paths, gt_paths

Get paths to the HVDROPDB data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • structure: The choice of anatomical structure to segment.
  • device: The choice of imaging device used to acquire the fundus images.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_hvdropdb_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], structure: Literal['optic_disc', 'vessels', 'ridge'], device: Literal['RetCam', 'Neo'] = 'RetCam', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
122def get_hvdropdb_dataset(
123    path: Union[os.PathLike, str],
124    patch_shape: Tuple[int, int],
125    structure: Literal["optic_disc", "vessels", "ridge"],
126    device: Literal["RetCam", "Neo"] = "RetCam",
127    resize_inputs: bool = False,
128    download: bool = False,
129    **kwargs
130) -> Dataset:
131    """Get the HVDROPDB dataset for optic disc / vessel / demarcation line segmentation in fundus images.
132
133    Args:
134        path: Filepath to a folder where the downloaded data will be saved.
135        patch_shape: The patch shape to use for training.
136        structure: The choice of anatomical structure to segment.
137        device: The choice of imaging device used to acquire the fundus images.
138        resize_inputs: Whether to resize the inputs to the expected patch shape.
139        download: Whether to download the data if it is not present.
140        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
141
142    Returns:
143        The segmentation dataset.
144    """
145    image_paths, gt_paths = get_hvdropdb_paths(path=path, structure=structure, device=device, download=download)
146
147    if resize_inputs:
148        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
149        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
150            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs,
151        )
152
153    return torch_em.default_segmentation_dataset(
154        raw_paths=image_paths,
155        raw_key=None,
156        label_paths=gt_paths,
157        label_key=None,
158        patch_shape=patch_shape,
159        is_seg_dataset=False,
160        **kwargs
161    )

Get the HVDROPDB dataset for optic disc / vessel / demarcation line segmentation in fundus images.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • patch_shape: The patch shape to use for training.
  • structure: The choice of anatomical structure to segment.
  • device: The choice of imaging device used to acquire the fundus images.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_hvdropdb_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], structure: Literal['optic_disc', 'vessels', 'ridge'], device: Literal['RetCam', 'Neo'] = 'RetCam', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
164def get_hvdropdb_loader(
165    path: Union[os.PathLike, str],
166    batch_size: int,
167    patch_shape: Tuple[int, int],
168    structure: Literal["optic_disc", "vessels", "ridge"],
169    device: Literal["RetCam", "Neo"] = "RetCam",
170    resize_inputs: bool = False,
171    download: bool = False,
172    **kwargs
173) -> DataLoader:
174    """Get the HVDROPDB dataloader for optic disc / vessel / demarcation line segmentation in fundus images.
175
176    Args:
177        path: Filepath to a folder where the downloaded data will be saved.
178        batch_size: The batch size for training.
179        patch_shape: The patch shape to use for training.
180        structure: The choice of anatomical structure to segment.
181        device: The choice of imaging device used to acquire the fundus images.
182        resize_inputs: Whether to resize the inputs to the expected patch shape.
183        download: Whether to download the data if it is not present.
184        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
185
186    Returns:
187        The DataLoader.
188    """
189    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
190    dataset = get_hvdropdb_dataset(path, patch_shape, structure, device, resize_inputs, download, **ds_kwargs)
191    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the HVDROPDB dataloader for optic disc / vessel / demarcation line segmentation in fundus images.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • structure: The choice of anatomical structure to segment.
  • device: The choice of imaging device used to acquire the fundus images.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.