torch_em.data.datasets.medical.chase_db1
The CHASE_DB1 dataset contains annotations for retinal vessel segmentation in fundus images.
This dataset is located at https://researchdata.kingston.ac.uk/96/. The dataset is from the publication https://doi.org/10.1109/TBME.2012.2205687. The dataset is licensed under CC BY 4.0 (see https://researchdata.kingston.ac.uk/96/ for details). Please cite the publication above if you use this dataset for your research.
1"""The CHASE_DB1 dataset contains annotations for retinal vessel segmentation 2in fundus images. 3 4This dataset is located at https://researchdata.kingston.ac.uk/96/. 5The dataset is from the publication https://doi.org/10.1109/TBME.2012.2205687. 6The dataset is licensed under CC BY 4.0 (see https://researchdata.kingston.ac.uk/96/ 7for details). Please cite the publication above if you use this dataset for your research. 8""" 9 10import os 11from glob import glob 12from typing import Union, Tuple, Literal, List 13 14from torch.utils.data import Dataset, DataLoader 15 16import torch_em 17 18from .. import util 19 20 21URL = "https://researchdata.kingston.ac.uk/96/1/CHASEDB1.zip" 22CHECKSUM = None 23 24 25def get_chase_db1_data(path: Union[os.PathLike, str], download: bool = False) -> str: 26 """Download the CHASE_DB1 dataset. 27 28 Args: 29 path: Filepath to a folder where the data is downloaded for further processing. 30 download: Whether to download the data if it is not present. 31 32 Returns: 33 Filepath where the data is downloaded. 34 """ 35 data_dir = os.path.join(path, "CHASEDB1") 36 if os.path.exists(data_dir): 37 return data_dir 38 39 os.makedirs(path, exist_ok=True) 40 41 zip_path = os.path.join(path, "CHASEDB1.zip") 42 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 43 util.unzip(zip_path=zip_path, dst=data_dir) 44 45 return data_dir 46 47 48def get_chase_db1_paths( 49 path: Union[os.PathLike, str], 50 split: Literal['train', 'val', 'test'], 51 annotator: Literal['1st', '2nd'] = '1st', 52 download: bool = False, 53) -> Tuple[List[str], List[str]]: 54 """Get paths to the CHASE_DB1 data. 55 56 Args: 57 path: Filepath to a folder where the data is downloaded for further processing. 58 split: The choice of data split. The dataset does not ship an official split. We use the first 59 20 images for training (of which the last 4 are held out for validation) and the 60 remaining 8 images for testing, following the split convention used in the vessel 61 segmentation literature. 62 annotator: The choice of annotator for the ground-truth vessel maps. There are two independent 63 manual annotations ('1st' and '2nd') per image. 64 download: Whether to download the data if it is not present. 65 66 Returns: 67 List of filepaths for the image data. 68 List of filepaths for the label data. 69 """ 70 data_dir = get_chase_db1_data(path=path, download=download) 71 72 assert annotator in ["1st", "2nd"], f"'{annotator}' is not a valid annotator choice." 73 74 image_paths = sorted(glob(os.path.join(data_dir, "Image_*.jpg"))) 75 gt_paths = sorted(glob(os.path.join(data_dir, f"Image_*_{annotator}HO.png"))) 76 77 assert len(image_paths) == len(gt_paths) and len(image_paths) > 0 78 79 if split == "train": 80 image_paths, gt_paths = image_paths[:16], gt_paths[:16] 81 elif split == "val": 82 image_paths, gt_paths = image_paths[16:20], gt_paths[16:20] 83 elif split == "test": 84 image_paths, gt_paths = image_paths[20:], gt_paths[20:] 85 else: 86 raise ValueError(f"'{split}' is not a valid split.") 87 88 return image_paths, gt_paths 89 90 91def get_chase_db1_dataset( 92 path: Union[os.PathLike, str], 93 patch_shape: Tuple[int, int], 94 split: Literal['train', 'val', 'test'], 95 annotator: Literal['1st', '2nd'] = '1st', 96 resize_inputs: bool = False, 97 download: bool = False, 98 **kwargs 99) -> Dataset: 100 """Get the CHASE_DB1 dataset for segmentation of retinal blood vessels in fundus images. 101 102 Args: 103 path: Filepath to a folder where the data is downloaded for further processing. 104 patch_shape: The patch shape to use for training. 105 split: The choice of data split. 106 annotator: The choice of annotator for the ground-truth vessel maps. 107 resize_inputs: Whether to resize the inputs to the expected patch shape. 108 download: Whether to download the data if it is not present. 109 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 110 111 Returns: 112 The segmentation dataset. 113 """ 114 image_paths, gt_paths = get_chase_db1_paths(path=path, split=split, annotator=annotator, download=download) 115 116 if resize_inputs: 117 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 118 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 119 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 120 ) 121 122 return torch_em.default_segmentation_dataset( 123 raw_paths=image_paths, 124 raw_key=None, 125 label_paths=gt_paths, 126 label_key=None, 127 patch_shape=patch_shape, 128 is_seg_dataset=False, 129 **kwargs 130 ) 131 132 133def get_chase_db1_loader( 134 path: Union[os.PathLike, str], 135 batch_size: int, 136 patch_shape: Tuple[int, int], 137 split: Literal['train', 'val', 'test'], 138 annotator: Literal['1st', '2nd'] = '1st', 139 resize_inputs: bool = False, 140 download: bool = False, 141 **kwargs 142) -> DataLoader: 143 """Get the CHASE_DB1 dataloader for segmentation of retinal blood vessels in fundus images. 144 145 Args: 146 path: Filepath to a folder where the data is downloaded for further processing. 147 batch_size: The batch size for training. 148 patch_shape: The patch shape to use for training. 149 split: The choice of data split. 150 annotator: The choice of annotator for the ground-truth vessel maps. 151 resize_inputs: Whether to resize the inputs to the expected patch shape. 152 download: Whether to download the data if it is not present. 153 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 154 155 Returns: 156 The DataLoader. 157 """ 158 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 159 dataset = get_chase_db1_dataset(path, patch_shape, split, annotator, resize_inputs, download, **ds_kwargs) 160 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
26def get_chase_db1_data(path: Union[os.PathLike, str], download: bool = False) -> str: 27 """Download the CHASE_DB1 dataset. 28 29 Args: 30 path: Filepath to a folder where the data is downloaded for further processing. 31 download: Whether to download the data if it is not present. 32 33 Returns: 34 Filepath where the data is downloaded. 35 """ 36 data_dir = os.path.join(path, "CHASEDB1") 37 if os.path.exists(data_dir): 38 return data_dir 39 40 os.makedirs(path, exist_ok=True) 41 42 zip_path = os.path.join(path, "CHASEDB1.zip") 43 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 44 util.unzip(zip_path=zip_path, dst=data_dir) 45 46 return data_dir
Download the CHASE_DB1 dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
49def get_chase_db1_paths( 50 path: Union[os.PathLike, str], 51 split: Literal['train', 'val', 'test'], 52 annotator: Literal['1st', '2nd'] = '1st', 53 download: bool = False, 54) -> Tuple[List[str], List[str]]: 55 """Get paths to the CHASE_DB1 data. 56 57 Args: 58 path: Filepath to a folder where the data is downloaded for further processing. 59 split: The choice of data split. The dataset does not ship an official split. We use the first 60 20 images for training (of which the last 4 are held out for validation) and the 61 remaining 8 images for testing, following the split convention used in the vessel 62 segmentation literature. 63 annotator: The choice of annotator for the ground-truth vessel maps. There are two independent 64 manual annotations ('1st' and '2nd') per image. 65 download: Whether to download the data if it is not present. 66 67 Returns: 68 List of filepaths for the image data. 69 List of filepaths for the label data. 70 """ 71 data_dir = get_chase_db1_data(path=path, download=download) 72 73 assert annotator in ["1st", "2nd"], f"'{annotator}' is not a valid annotator choice." 74 75 image_paths = sorted(glob(os.path.join(data_dir, "Image_*.jpg"))) 76 gt_paths = sorted(glob(os.path.join(data_dir, f"Image_*_{annotator}HO.png"))) 77 78 assert len(image_paths) == len(gt_paths) and len(image_paths) > 0 79 80 if split == "train": 81 image_paths, gt_paths = image_paths[:16], gt_paths[:16] 82 elif split == "val": 83 image_paths, gt_paths = image_paths[16:20], gt_paths[16:20] 84 elif split == "test": 85 image_paths, gt_paths = image_paths[20:], gt_paths[20:] 86 else: 87 raise ValueError(f"'{split}' is not a valid split.") 88 89 return image_paths, gt_paths
Get paths to the CHASE_DB1 data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- split: The choice of data split. The dataset does not ship an official split. We use the first 20 images for training (of which the last 4 are held out for validation) and the remaining 8 images for testing, following the split convention used in the vessel segmentation literature.
- annotator: The choice of annotator for the ground-truth vessel maps. There are two independent manual annotations ('1st' and '2nd') per image.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
92def get_chase_db1_dataset( 93 path: Union[os.PathLike, str], 94 patch_shape: Tuple[int, int], 95 split: Literal['train', 'val', 'test'], 96 annotator: Literal['1st', '2nd'] = '1st', 97 resize_inputs: bool = False, 98 download: bool = False, 99 **kwargs 100) -> Dataset: 101 """Get the CHASE_DB1 dataset for segmentation of retinal blood vessels in fundus images. 102 103 Args: 104 path: Filepath to a folder where the data is downloaded for further processing. 105 patch_shape: The patch shape to use for training. 106 split: The choice of data split. 107 annotator: The choice of annotator for the ground-truth vessel maps. 108 resize_inputs: Whether to resize the inputs to the expected patch shape. 109 download: Whether to download the data if it is not present. 110 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 111 112 Returns: 113 The segmentation dataset. 114 """ 115 image_paths, gt_paths = get_chase_db1_paths(path=path, split=split, annotator=annotator, download=download) 116 117 if resize_inputs: 118 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 119 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 120 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 121 ) 122 123 return torch_em.default_segmentation_dataset( 124 raw_paths=image_paths, 125 raw_key=None, 126 label_paths=gt_paths, 127 label_key=None, 128 patch_shape=patch_shape, 129 is_seg_dataset=False, 130 **kwargs 131 )
Get the CHASE_DB1 dataset for segmentation of retinal blood vessels in fundus images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- split: The choice of data split.
- annotator: The choice of annotator for the ground-truth vessel maps.
- resize_inputs: Whether to resize the inputs to the expected patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
134def get_chase_db1_loader( 135 path: Union[os.PathLike, str], 136 batch_size: int, 137 patch_shape: Tuple[int, int], 138 split: Literal['train', 'val', 'test'], 139 annotator: Literal['1st', '2nd'] = '1st', 140 resize_inputs: bool = False, 141 download: bool = False, 142 **kwargs 143) -> DataLoader: 144 """Get the CHASE_DB1 dataloader for segmentation of retinal blood vessels in fundus images. 145 146 Args: 147 path: Filepath to a folder where the data is downloaded for further processing. 148 batch_size: The batch size for training. 149 patch_shape: The patch shape to use for training. 150 split: The choice of data split. 151 annotator: The choice of annotator for the ground-truth vessel maps. 152 resize_inputs: Whether to resize the inputs to the expected patch shape. 153 download: Whether to download the data if it is not present. 154 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 155 156 Returns: 157 The DataLoader. 158 """ 159 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 160 dataset = get_chase_db1_dataset(path, patch_shape, split, annotator, resize_inputs, download, **ds_kwargs) 161 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the CHASE_DB1 dataloader for segmentation of retinal blood vessels in fundus images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- split: The choice of data split.
- annotator: The choice of annotator for the ground-truth vessel maps.
- resize_inputs: Whether to resize the inputs to the expected patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.