torch_em.data.datasets.medical.stare
The STARE dataset contains annotations for retinal vessel segmentation in fundus images.
STARE (STructured Analysis of the Retina) is located at http://cecas.clemson.edu/~ahoover/stare/. It provides 20 fundus images along with two independent sets of hand-labeled vessel maps, created by Adam Hoover ('ah') and Valentina Kouznetsova ('vk').
The dataset is from the publication https://doi.org/10.1109/42.845178. Please cite it if you use this dataset for your research.
1"""The STARE dataset contains annotations for retinal vessel segmentation in fundus images. 2 3STARE (STructured Analysis of the Retina) is located at http://cecas.clemson.edu/~ahoover/stare/. 4It provides 20 fundus images along with two independent sets of hand-labeled vessel maps, created 5by Adam Hoover ('ah') and Valentina Kouznetsova ('vk'). 6 7The dataset is from the publication https://doi.org/10.1109/42.845178. 8Please cite it if you use this dataset for your research. 9""" 10 11import os 12import gzip 13import shutil 14from glob import glob 15from typing import Union, Tuple, Literal, List 16 17from torch.utils.data import Dataset, DataLoader 18 19import torch_em 20 21from .. import util 22 23 24URL = { 25 "images": "http://cecas.clemson.edu/~ahoover/stare/probing/stare-images.tar", 26 "ah": "http://cecas.clemson.edu/~ahoover/stare/probing/labels-ah.tar", 27 "vk": "http://cecas.clemson.edu/~ahoover/stare/probing/labels-vk.tar", 28} 29 30CHECKSUM = { 31 "images": "5f7b509b6067cad4f1be84933145d783de9ec087b3eaaf4db0103dd0144dd433", 32 "ah": "ebf2f1e17ca955f24579d9edd990e2dae79a5c82def69f0985d8e24f826ddd2f", 33 "vk": "47474a701536b0cfdb369fdce012be36141e9f44d80387f0179446b5cb0f5576", 34} 35 36 37def _unpack_gz_files(dir_path): 38 gz_paths = sorted(glob(os.path.join(dir_path, "*.gz"))) 39 for gz_path in gz_paths: 40 out_path = gz_path[:-len(".gz")] 41 if os.path.exists(out_path): 42 continue 43 with gzip.open(gz_path, "rb") as f_in, open(out_path, "wb") as f_out: 44 shutil.copyfileobj(f_in, f_out) 45 46 47def get_stare_data(path: Union[os.PathLike, str], download: bool = False) -> str: 48 """Download the STARE dataset. 49 50 Args: 51 path: Filepath to a folder where the data is downloaded for further processing. 52 download: Whether to download the data if it is not present. 53 54 Returns: 55 Filepath where the data is downloaded. 56 """ 57 images_dir = os.path.join(path, "images") 58 if os.path.exists(images_dir): 59 return path 60 61 os.makedirs(path, exist_ok=True) 62 63 for name, target_dir in [("images", "images"), ("ah", "labels-ah"), ("vk", "labels-vk")]: 64 tar_path = os.path.join(path, f"{name}.tar") 65 util.download_source(path=tar_path, url=URL[name], download=download, checksum=CHECKSUM[name]) 66 67 extract_dir = os.path.join(path, target_dir) 68 os.makedirs(extract_dir, exist_ok=True) 69 shutil.unpack_archive(tar_path, extract_dir, format="tar") 70 71 _unpack_gz_files(extract_dir) 72 73 return path 74 75 76def get_stare_paths( 77 path: Union[os.PathLike, str], 78 annotator: Literal["ah", "vk"] = "ah", 79 download: bool = False, 80) -> Tuple[List[str], List[str]]: 81 """Get paths to the STARE data. 82 83 Args: 84 path: Filepath to a folder where the data is downloaded for further processing. 85 annotator: The choice of annotator for the ground-truth vessel maps. There are two independent 86 manual annotations, provided by 'ah' (Adam Hoover) and 'vk' (Valentina Kouznetsova). 87 download: Whether to download the data if it is not present. 88 89 Returns: 90 List of filepaths for the image data. 91 List of filepaths for the label data. 92 """ 93 data_dir = get_stare_data(path=path, download=download) 94 95 assert annotator in ["ah", "vk"], f"'{annotator}' is not a valid annotator choice." 96 97 image_paths = sorted(glob(os.path.join(data_dir, "images", "*.ppm"))) 98 gt_paths = sorted(glob(os.path.join(data_dir, f"labels-{annotator}", f"*.{annotator}.ppm"))) 99 100 assert len(image_paths) == len(gt_paths) and len(image_paths) > 0 101 102 return image_paths, gt_paths 103 104 105def get_stare_dataset( 106 path: Union[os.PathLike, str], 107 patch_shape: Tuple[int, int], 108 annotator: Literal["ah", "vk"] = "ah", 109 resize_inputs: bool = False, 110 download: bool = False, 111 **kwargs 112) -> Dataset: 113 """Get the STARE dataset for segmentation of retinal blood vessels in fundus images. 114 115 Args: 116 path: Filepath to a folder where the data is downloaded for further processing. 117 patch_shape: The patch shape to use for training. 118 annotator: The choice of annotator for the ground-truth vessel maps. 119 resize_inputs: Whether to resize the inputs to the expected patch shape. 120 download: Whether to download the data if it is not present. 121 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 122 123 Returns: 124 The segmentation dataset. 125 """ 126 image_paths, gt_paths = get_stare_paths(path=path, annotator=annotator, download=download) 127 128 if resize_inputs: 129 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 130 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 131 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 132 ) 133 134 return torch_em.default_segmentation_dataset( 135 raw_paths=image_paths, 136 raw_key=None, 137 label_paths=gt_paths, 138 label_key=None, 139 patch_shape=patch_shape, 140 is_seg_dataset=False, 141 **kwargs 142 ) 143 144 145def get_stare_loader( 146 path: Union[os.PathLike, str], 147 batch_size: int, 148 patch_shape: Tuple[int, int], 149 annotator: Literal["ah", "vk"] = "ah", 150 resize_inputs: bool = False, 151 download: bool = False, 152 **kwargs 153) -> DataLoader: 154 """Get the STARE dataloader for segmentation of retinal blood vessels in fundus images. 155 156 Args: 157 path: Filepath to a folder where the data is downloaded for further processing. 158 batch_size: The batch size for training. 159 patch_shape: The patch shape to use for training. 160 annotator: The choice of annotator for the ground-truth vessel maps. 161 resize_inputs: Whether to resize the inputs to the expected patch shape. 162 download: Whether to download the data if it is not present. 163 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 164 165 Returns: 166 The DataLoader. 167 """ 168 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 169 dataset = get_stare_dataset(path, patch_shape, annotator, resize_inputs, download, **ds_kwargs) 170 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
48def get_stare_data(path: Union[os.PathLike, str], download: bool = False) -> str: 49 """Download the STARE dataset. 50 51 Args: 52 path: Filepath to a folder where the data is downloaded for further processing. 53 download: Whether to download the data if it is not present. 54 55 Returns: 56 Filepath where the data is downloaded. 57 """ 58 images_dir = os.path.join(path, "images") 59 if os.path.exists(images_dir): 60 return path 61 62 os.makedirs(path, exist_ok=True) 63 64 for name, target_dir in [("images", "images"), ("ah", "labels-ah"), ("vk", "labels-vk")]: 65 tar_path = os.path.join(path, f"{name}.tar") 66 util.download_source(path=tar_path, url=URL[name], download=download, checksum=CHECKSUM[name]) 67 68 extract_dir = os.path.join(path, target_dir) 69 os.makedirs(extract_dir, exist_ok=True) 70 shutil.unpack_archive(tar_path, extract_dir, format="tar") 71 72 _unpack_gz_files(extract_dir) 73 74 return path
Download the STARE dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
77def get_stare_paths( 78 path: Union[os.PathLike, str], 79 annotator: Literal["ah", "vk"] = "ah", 80 download: bool = False, 81) -> Tuple[List[str], List[str]]: 82 """Get paths to the STARE data. 83 84 Args: 85 path: Filepath to a folder where the data is downloaded for further processing. 86 annotator: The choice of annotator for the ground-truth vessel maps. There are two independent 87 manual annotations, provided by 'ah' (Adam Hoover) and 'vk' (Valentina Kouznetsova). 88 download: Whether to download the data if it is not present. 89 90 Returns: 91 List of filepaths for the image data. 92 List of filepaths for the label data. 93 """ 94 data_dir = get_stare_data(path=path, download=download) 95 96 assert annotator in ["ah", "vk"], f"'{annotator}' is not a valid annotator choice." 97 98 image_paths = sorted(glob(os.path.join(data_dir, "images", "*.ppm"))) 99 gt_paths = sorted(glob(os.path.join(data_dir, f"labels-{annotator}", f"*.{annotator}.ppm"))) 100 101 assert len(image_paths) == len(gt_paths) and len(image_paths) > 0 102 103 return image_paths, gt_paths
Get paths to the STARE data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- annotator: The choice of annotator for the ground-truth vessel maps. There are two independent manual annotations, provided by 'ah' (Adam Hoover) and 'vk' (Valentina Kouznetsova).
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
106def get_stare_dataset( 107 path: Union[os.PathLike, str], 108 patch_shape: Tuple[int, int], 109 annotator: Literal["ah", "vk"] = "ah", 110 resize_inputs: bool = False, 111 download: bool = False, 112 **kwargs 113) -> Dataset: 114 """Get the STARE dataset for segmentation of retinal blood vessels in fundus images. 115 116 Args: 117 path: Filepath to a folder where the data is downloaded for further processing. 118 patch_shape: The patch shape to use for training. 119 annotator: The choice of annotator for the ground-truth vessel maps. 120 resize_inputs: Whether to resize the inputs to the expected patch shape. 121 download: Whether to download the data if it is not present. 122 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 123 124 Returns: 125 The segmentation dataset. 126 """ 127 image_paths, gt_paths = get_stare_paths(path=path, annotator=annotator, download=download) 128 129 if resize_inputs: 130 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 131 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 132 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 133 ) 134 135 return torch_em.default_segmentation_dataset( 136 raw_paths=image_paths, 137 raw_key=None, 138 label_paths=gt_paths, 139 label_key=None, 140 patch_shape=patch_shape, 141 is_seg_dataset=False, 142 **kwargs 143 )
Get the STARE dataset for segmentation of retinal blood vessels in fundus images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- annotator: The choice of annotator for the ground-truth vessel maps.
- resize_inputs: Whether to resize the inputs to the expected patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
146def get_stare_loader( 147 path: Union[os.PathLike, str], 148 batch_size: int, 149 patch_shape: Tuple[int, int], 150 annotator: Literal["ah", "vk"] = "ah", 151 resize_inputs: bool = False, 152 download: bool = False, 153 **kwargs 154) -> DataLoader: 155 """Get the STARE dataloader for segmentation of retinal blood vessels in fundus images. 156 157 Args: 158 path: Filepath to a folder where the data is downloaded for further processing. 159 batch_size: The batch size for training. 160 patch_shape: The patch shape to use for training. 161 annotator: The choice of annotator for the ground-truth vessel maps. 162 resize_inputs: Whether to resize the inputs to the expected patch shape. 163 download: Whether to download the data if it is not present. 164 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 165 166 Returns: 167 The DataLoader. 168 """ 169 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 170 dataset = get_stare_dataset(path, patch_shape, annotator, resize_inputs, download, **ds_kwargs) 171 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the STARE dataloader for segmentation of retinal blood vessels in fundus images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- annotator: The choice of annotator for the ground-truth vessel maps.
- resize_inputs: Whether to resize the inputs to the expected patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.