torch_em.data.datasets.medical.hvdropdb
The HVDROPDB dataset contains annotations for optic disc, blood vessel, and demarcation line / ridge segmentation in fundus images of premature infants, for the task of retinopathy of prematurity (ROP) screening.
This dataset is located at https://doi.org/10.17632/xw5xc7xrmp.3, under the CC BY 4.0 license. The dataset is from the publication https://doi.org/10.1016/j.dib.2023.109839. Please cite it if you use this dataset for your research.
The images were acquired with two imaging systems (RetCam and Neo) at PBMA's H.V. Desai Eye Hospital, Pune, screening preterm infants for ROP. Ground truth masks were prepared manually (with Adobe Photoshop) by a group of ROP experts, separately for the optic disc, blood vessels, and the demarcation line / ridge.
1"""The HVDROPDB dataset contains annotations for optic disc, blood vessel, and 2demarcation line / ridge segmentation in fundus images of premature infants, 3for the task of retinopathy of prematurity (ROP) screening. 4 5This dataset is located at https://doi.org/10.17632/xw5xc7xrmp.3, under the CC BY 4.0 license. 6The dataset is from the publication https://doi.org/10.1016/j.dib.2023.109839. 7Please cite it if you use this dataset for your research. 8 9The images were acquired with two imaging systems (RetCam and Neo) at PBMA's H.V. Desai Eye 10Hospital, Pune, screening preterm infants for ROP. Ground truth masks were prepared manually 11(with Adobe Photoshop) by a group of ROP experts, separately for the optic disc, blood vessels, 12and the demarcation line / ridge. 13""" 14 15import os 16from glob import glob 17from pathlib import Path 18from natsort import natsorted 19from typing import Union, Tuple, Literal, List 20 21import imageio.v3 as imageio 22 23from torch.utils.data import Dataset, DataLoader 24 25import torch_em 26 27from .. import util 28 29 30URL = "https://data.mendeley.com/public-files/datasets/xw5xc7xrmp/files/dfa3ec48-c763-439f-905a-bb0904ea840a/file_downloaded" # noqa 31CHECKSUM = "0a3aa96bbe489fc2f9103bac80fdb3eddc4fc91533e87555446970e9d021c1d2" 32 33# Maps the structure to segment to the folder name and the image / mask subfolder prefixes 34# used per imaging device inside the archive. 35STRUCTURES = { 36 "optic_disc": ("HVDROPDB-OD", {"RetCam": "Retcam_OpticDisc", "Neo": "Neo_OpticDisc"}), 37 "vessels": ("HVDROPDB-BV", {"RetCam": "RetCam_Vessels", "Neo": "Neo_Vessels"}), 38 "ridge": ("HVDROPDB-RIDGE", {"RetCam": "RetCam_Ridge", "Neo": "Neo_Ridge"}), 39} 40 41 42def get_hvdropdb_data(path: Union[os.PathLike, str], download: bool = False) -> str: 43 """Download the HVDROPDB dataset. 44 45 Args: 46 path: Filepath to a folder where the data is downloaded for further processing. 47 download: Whether to download the data if it is not present. 48 49 Returns: 50 Filepath where the data is downloaded. 51 """ 52 data_dir = os.path.join(path, "HVDROPDB_RetCam_Neo_Segmentation") 53 if os.path.exists(data_dir): 54 return data_dir 55 56 os.makedirs(path, exist_ok=True) 57 58 zip_path = os.path.join(path, "HVDROPDB_RetCam_Neo_Segmentation.zip") 59 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 60 util.unzip(zip_path=zip_path, dst=path) 61 62 return data_dir 63 64 65def _binarize_mask(mask_path, dst_dir): 66 dst_path = os.path.join(dst_dir, Path(mask_path).stem + ".png") 67 if os.path.exists(dst_path): 68 return dst_path 69 70 os.makedirs(dst_dir, exist_ok=True) 71 mask = imageio.imread(mask_path) 72 if mask.ndim == 3: 73 mask = mask[..., 0] 74 mask = (mask > 127).astype("uint8") 75 imageio.imwrite(dst_path, mask) 76 return dst_path 77 78 79def get_hvdropdb_paths( 80 path: Union[os.PathLike, str], 81 structure: Literal["optic_disc", "vessels", "ridge"], 82 device: Literal["RetCam", "Neo"] = "RetCam", 83 download: bool = False, 84) -> Tuple[List[str], List[str]]: 85 """Get paths to the HVDROPDB data. 86 87 Args: 88 path: Filepath to a folder where the data is downloaded for further processing. 89 structure: The choice of anatomical structure to segment. 90 device: The choice of imaging device used to acquire the fundus images. 91 download: Whether to download the data if it is not present. 92 93 Returns: 94 List of filepaths for the image data. 95 List of filepaths for the label data. 96 """ 97 assert structure in STRUCTURES, f"'{structure}' is not a valid structure choice." 98 assert device in ["RetCam", "Neo"], f"'{device}' is not a valid device choice." 99 100 data_dir = get_hvdropdb_data(path=path, download=download) 101 102 folder, prefixes = STRUCTURES[structure] 103 prefix = prefixes[device] 104 105 image_dir = os.path.join(data_dir, folder, f"{prefix}_images") 106 gt_dir = os.path.join(data_dir, folder, f"{prefix}_masks") 107 108 image_paths = natsorted(glob(os.path.join(image_dir, "*.png"))) 109 raw_gt_paths = natsorted(glob(os.path.join(gt_dir, "*.png"))) 110 111 assert len(image_paths) == len(raw_gt_paths) and len(image_paths) > 0 112 113 # The shipped masks are RGB images (with an identical value per channel). We binarize 114 # them into single-channel masks here, as expected by `torch_em.default_segmentation_dataset`. 115 binary_gt_dir = os.path.join(data_dir, folder, f"{prefix}_masks_binary") 116 gt_paths = [_binarize_mask(p, binary_gt_dir) for p in raw_gt_paths] 117 118 return image_paths, gt_paths 119 120 121def get_hvdropdb_dataset( 122 path: Union[os.PathLike, str], 123 patch_shape: Tuple[int, int], 124 structure: Literal["optic_disc", "vessels", "ridge"], 125 device: Literal["RetCam", "Neo"] = "RetCam", 126 resize_inputs: bool = False, 127 download: bool = False, 128 **kwargs 129) -> Dataset: 130 """Get the HVDROPDB dataset for optic disc / vessel / demarcation line segmentation in fundus images. 131 132 Args: 133 path: Filepath to a folder where the downloaded data will be saved. 134 patch_shape: The patch shape to use for training. 135 structure: The choice of anatomical structure to segment. 136 device: The choice of imaging device used to acquire the fundus images. 137 resize_inputs: Whether to resize the inputs to the expected patch shape. 138 download: Whether to download the data if it is not present. 139 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 140 141 Returns: 142 The segmentation dataset. 143 """ 144 image_paths, gt_paths = get_hvdropdb_paths(path=path, structure=structure, device=device, download=download) 145 146 if resize_inputs: 147 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 148 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 149 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs, 150 ) 151 152 return torch_em.default_segmentation_dataset( 153 raw_paths=image_paths, 154 raw_key=None, 155 label_paths=gt_paths, 156 label_key=None, 157 patch_shape=patch_shape, 158 is_seg_dataset=False, 159 **kwargs 160 ) 161 162 163def get_hvdropdb_loader( 164 path: Union[os.PathLike, str], 165 batch_size: int, 166 patch_shape: Tuple[int, int], 167 structure: Literal["optic_disc", "vessels", "ridge"], 168 device: Literal["RetCam", "Neo"] = "RetCam", 169 resize_inputs: bool = False, 170 download: bool = False, 171 **kwargs 172) -> DataLoader: 173 """Get the HVDROPDB dataloader for optic disc / vessel / demarcation line segmentation in fundus images. 174 175 Args: 176 path: Filepath to a folder where the downloaded data will be saved. 177 batch_size: The batch size for training. 178 patch_shape: The patch shape to use for training. 179 structure: The choice of anatomical structure to segment. 180 device: The choice of imaging device used to acquire the fundus images. 181 resize_inputs: Whether to resize the inputs to the expected patch shape. 182 download: Whether to download the data if it is not present. 183 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 184 185 Returns: 186 The DataLoader. 187 """ 188 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 189 dataset = get_hvdropdb_dataset(path, patch_shape, structure, device, resize_inputs, download, **ds_kwargs) 190 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
43def get_hvdropdb_data(path: Union[os.PathLike, str], download: bool = False) -> str: 44 """Download the HVDROPDB dataset. 45 46 Args: 47 path: Filepath to a folder where the data is downloaded for further processing. 48 download: Whether to download the data if it is not present. 49 50 Returns: 51 Filepath where the data is downloaded. 52 """ 53 data_dir = os.path.join(path, "HVDROPDB_RetCam_Neo_Segmentation") 54 if os.path.exists(data_dir): 55 return data_dir 56 57 os.makedirs(path, exist_ok=True) 58 59 zip_path = os.path.join(path, "HVDROPDB_RetCam_Neo_Segmentation.zip") 60 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 61 util.unzip(zip_path=zip_path, dst=path) 62 63 return data_dir
Download the HVDROPDB dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
80def get_hvdropdb_paths( 81 path: Union[os.PathLike, str], 82 structure: Literal["optic_disc", "vessels", "ridge"], 83 device: Literal["RetCam", "Neo"] = "RetCam", 84 download: bool = False, 85) -> Tuple[List[str], List[str]]: 86 """Get paths to the HVDROPDB data. 87 88 Args: 89 path: Filepath to a folder where the data is downloaded for further processing. 90 structure: The choice of anatomical structure to segment. 91 device: The choice of imaging device used to acquire the fundus images. 92 download: Whether to download the data if it is not present. 93 94 Returns: 95 List of filepaths for the image data. 96 List of filepaths for the label data. 97 """ 98 assert structure in STRUCTURES, f"'{structure}' is not a valid structure choice." 99 assert device in ["RetCam", "Neo"], f"'{device}' is not a valid device choice." 100 101 data_dir = get_hvdropdb_data(path=path, download=download) 102 103 folder, prefixes = STRUCTURES[structure] 104 prefix = prefixes[device] 105 106 image_dir = os.path.join(data_dir, folder, f"{prefix}_images") 107 gt_dir = os.path.join(data_dir, folder, f"{prefix}_masks") 108 109 image_paths = natsorted(glob(os.path.join(image_dir, "*.png"))) 110 raw_gt_paths = natsorted(glob(os.path.join(gt_dir, "*.png"))) 111 112 assert len(image_paths) == len(raw_gt_paths) and len(image_paths) > 0 113 114 # The shipped masks are RGB images (with an identical value per channel). We binarize 115 # them into single-channel masks here, as expected by `torch_em.default_segmentation_dataset`. 116 binary_gt_dir = os.path.join(data_dir, folder, f"{prefix}_masks_binary") 117 gt_paths = [_binarize_mask(p, binary_gt_dir) for p in raw_gt_paths] 118 119 return image_paths, gt_paths
Get paths to the HVDROPDB data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- structure: The choice of anatomical structure to segment.
- device: The choice of imaging device used to acquire the fundus images.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
122def get_hvdropdb_dataset( 123 path: Union[os.PathLike, str], 124 patch_shape: Tuple[int, int], 125 structure: Literal["optic_disc", "vessels", "ridge"], 126 device: Literal["RetCam", "Neo"] = "RetCam", 127 resize_inputs: bool = False, 128 download: bool = False, 129 **kwargs 130) -> Dataset: 131 """Get the HVDROPDB dataset for optic disc / vessel / demarcation line segmentation in fundus images. 132 133 Args: 134 path: Filepath to a folder where the downloaded data will be saved. 135 patch_shape: The patch shape to use for training. 136 structure: The choice of anatomical structure to segment. 137 device: The choice of imaging device used to acquire the fundus images. 138 resize_inputs: Whether to resize the inputs to the expected patch shape. 139 download: Whether to download the data if it is not present. 140 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 141 142 Returns: 143 The segmentation dataset. 144 """ 145 image_paths, gt_paths = get_hvdropdb_paths(path=path, structure=structure, device=device, download=download) 146 147 if resize_inputs: 148 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 149 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 150 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs, 151 ) 152 153 return torch_em.default_segmentation_dataset( 154 raw_paths=image_paths, 155 raw_key=None, 156 label_paths=gt_paths, 157 label_key=None, 158 patch_shape=patch_shape, 159 is_seg_dataset=False, 160 **kwargs 161 )
Get the HVDROPDB dataset for optic disc / vessel / demarcation line segmentation in fundus images.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- patch_shape: The patch shape to use for training.
- structure: The choice of anatomical structure to segment.
- device: The choice of imaging device used to acquire the fundus images.
- resize_inputs: Whether to resize the inputs to the expected patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
164def get_hvdropdb_loader( 165 path: Union[os.PathLike, str], 166 batch_size: int, 167 patch_shape: Tuple[int, int], 168 structure: Literal["optic_disc", "vessels", "ridge"], 169 device: Literal["RetCam", "Neo"] = "RetCam", 170 resize_inputs: bool = False, 171 download: bool = False, 172 **kwargs 173) -> DataLoader: 174 """Get the HVDROPDB dataloader for optic disc / vessel / demarcation line segmentation in fundus images. 175 176 Args: 177 path: Filepath to a folder where the downloaded data will be saved. 178 batch_size: The batch size for training. 179 patch_shape: The patch shape to use for training. 180 structure: The choice of anatomical structure to segment. 181 device: The choice of imaging device used to acquire the fundus images. 182 resize_inputs: Whether to resize the inputs to the expected patch shape. 183 download: Whether to download the data if it is not present. 184 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 185 186 Returns: 187 The DataLoader. 188 """ 189 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 190 dataset = get_hvdropdb_dataset(path, patch_shape, structure, device, resize_inputs, download, **ds_kwargs) 191 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the HVDROPDB dataloader for optic disc / vessel / demarcation line segmentation in fundus images.
Arguments:
- path: Filepath to a folder where the downloaded data will be saved.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- structure: The choice of anatomical structure to segment.
- device: The choice of imaging device used to acquire the fundus images.
- resize_inputs: Whether to resize the inputs to the expected patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.