torch_em.data.datasets.medical.chase_db1

The CHASE_DB1 dataset contains annotations for retinal vessel segmentation in fundus images.

This dataset is located at https://researchdata.kingston.ac.uk/96/. The dataset is from the publication https://doi.org/10.1109/TBME.2012.2205687. The dataset is licensed under CC BY 4.0 (see https://researchdata.kingston.ac.uk/96/ for details). Please cite the publication above if you use this dataset for your research.

  1"""The CHASE_DB1 dataset contains annotations for retinal vessel segmentation
  2in fundus images.
  3
  4This dataset is located at https://researchdata.kingston.ac.uk/96/.
  5The dataset is from the publication https://doi.org/10.1109/TBME.2012.2205687.
  6The dataset is licensed under CC BY 4.0 (see https://researchdata.kingston.ac.uk/96/
  7for details). Please cite the publication above if you use this dataset for your research.
  8"""
  9
 10import os
 11from glob import glob
 12from typing import Union, Tuple, Literal, List
 13
 14from torch.utils.data import Dataset, DataLoader
 15
 16import torch_em
 17
 18from .. import util
 19
 20
 21URL = "https://researchdata.kingston.ac.uk/96/1/CHASEDB1.zip"
 22CHECKSUM = None
 23
 24
 25def get_chase_db1_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 26    """Download the CHASE_DB1 dataset.
 27
 28    Args:
 29        path: Filepath to a folder where the data is downloaded for further processing.
 30        download: Whether to download the data if it is not present.
 31
 32    Returns:
 33        Filepath where the data is downloaded.
 34    """
 35    data_dir = os.path.join(path, "CHASEDB1")
 36    if os.path.exists(data_dir):
 37        return data_dir
 38
 39    os.makedirs(path, exist_ok=True)
 40
 41    zip_path = os.path.join(path, "CHASEDB1.zip")
 42    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 43    util.unzip(zip_path=zip_path, dst=data_dir)
 44
 45    return data_dir
 46
 47
 48def get_chase_db1_paths(
 49    path: Union[os.PathLike, str],
 50    split: Literal['train', 'val', 'test'],
 51    annotator: Literal['1st', '2nd'] = '1st',
 52    download: bool = False,
 53) -> Tuple[List[str], List[str]]:
 54    """Get paths to the CHASE_DB1 data.
 55
 56    Args:
 57        path: Filepath to a folder where the data is downloaded for further processing.
 58        split: The choice of data split. The dataset does not ship an official split. We use the first
 59            20 images for training (of which the last 4 are held out for validation) and the
 60            remaining 8 images for testing, following the split convention used in the vessel
 61            segmentation literature.
 62        annotator: The choice of annotator for the ground-truth vessel maps. There are two independent
 63            manual annotations ('1st' and '2nd') per image.
 64        download: Whether to download the data if it is not present.
 65
 66    Returns:
 67        List of filepaths for the image data.
 68        List of filepaths for the label data.
 69    """
 70    data_dir = get_chase_db1_data(path=path, download=download)
 71
 72    assert annotator in ["1st", "2nd"], f"'{annotator}' is not a valid annotator choice."
 73
 74    image_paths = sorted(glob(os.path.join(data_dir, "Image_*.jpg")))
 75    gt_paths = sorted(glob(os.path.join(data_dir, f"Image_*_{annotator}HO.png")))
 76
 77    assert len(image_paths) == len(gt_paths) and len(image_paths) > 0
 78
 79    if split == "train":
 80        image_paths, gt_paths = image_paths[:16], gt_paths[:16]
 81    elif split == "val":
 82        image_paths, gt_paths = image_paths[16:20], gt_paths[16:20]
 83    elif split == "test":
 84        image_paths, gt_paths = image_paths[20:], gt_paths[20:]
 85    else:
 86        raise ValueError(f"'{split}' is not a valid split.")
 87
 88    return image_paths, gt_paths
 89
 90
 91def get_chase_db1_dataset(
 92    path: Union[os.PathLike, str],
 93    patch_shape: Tuple[int, int],
 94    split: Literal['train', 'val', 'test'],
 95    annotator: Literal['1st', '2nd'] = '1st',
 96    resize_inputs: bool = False,
 97    download: bool = False,
 98    **kwargs
 99) -> Dataset:
100    """Get the CHASE_DB1 dataset for segmentation of retinal blood vessels in fundus images.
101
102    Args:
103        path: Filepath to a folder where the data is downloaded for further processing.
104        patch_shape: The patch shape to use for training.
105        split: The choice of data split.
106        annotator: The choice of annotator for the ground-truth vessel maps.
107        resize_inputs: Whether to resize the inputs to the expected patch shape.
108        download: Whether to download the data if it is not present.
109        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
110
111    Returns:
112        The segmentation dataset.
113    """
114    image_paths, gt_paths = get_chase_db1_paths(path=path, split=split, annotator=annotator, download=download)
115
116    if resize_inputs:
117        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
118        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
119            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
120        )
121
122    return torch_em.default_segmentation_dataset(
123        raw_paths=image_paths,
124        raw_key=None,
125        label_paths=gt_paths,
126        label_key=None,
127        patch_shape=patch_shape,
128        is_seg_dataset=False,
129        **kwargs
130    )
131
132
133def get_chase_db1_loader(
134    path: Union[os.PathLike, str],
135    batch_size: int,
136    patch_shape: Tuple[int, int],
137    split: Literal['train', 'val', 'test'],
138    annotator: Literal['1st', '2nd'] = '1st',
139    resize_inputs: bool = False,
140    download: bool = False,
141    **kwargs
142) -> DataLoader:
143    """Get the CHASE_DB1 dataloader for segmentation of retinal blood vessels in fundus images.
144
145    Args:
146        path: Filepath to a folder where the data is downloaded for further processing.
147        batch_size: The batch size for training.
148        patch_shape: The patch shape to use for training.
149        split: The choice of data split.
150        annotator: The choice of annotator for the ground-truth vessel maps.
151        resize_inputs: Whether to resize the inputs to the expected patch shape.
152        download: Whether to download the data if it is not present.
153        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
154
155    Returns:
156        The DataLoader.
157    """
158    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
159    dataset = get_chase_db1_dataset(path, patch_shape, split, annotator, resize_inputs, download, **ds_kwargs)
160    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://researchdata.kingston.ac.uk/96/1/CHASEDB1.zip'
CHECKSUM = None
def get_chase_db1_data(path: Union[os.PathLike, str], download: bool = False) -> str:
26def get_chase_db1_data(path: Union[os.PathLike, str], download: bool = False) -> str:
27    """Download the CHASE_DB1 dataset.
28
29    Args:
30        path: Filepath to a folder where the data is downloaded for further processing.
31        download: Whether to download the data if it is not present.
32
33    Returns:
34        Filepath where the data is downloaded.
35    """
36    data_dir = os.path.join(path, "CHASEDB1")
37    if os.path.exists(data_dir):
38        return data_dir
39
40    os.makedirs(path, exist_ok=True)
41
42    zip_path = os.path.join(path, "CHASEDB1.zip")
43    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
44    util.unzip(zip_path=zip_path, dst=data_dir)
45
46    return data_dir

Download the CHASE_DB1 dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_chase_db1_paths( path: Union[os.PathLike, str], split: Literal['train', 'val', 'test'], annotator: Literal['1st', '2nd'] = '1st', download: bool = False) -> Tuple[List[str], List[str]]:
49def get_chase_db1_paths(
50    path: Union[os.PathLike, str],
51    split: Literal['train', 'val', 'test'],
52    annotator: Literal['1st', '2nd'] = '1st',
53    download: bool = False,
54) -> Tuple[List[str], List[str]]:
55    """Get paths to the CHASE_DB1 data.
56
57    Args:
58        path: Filepath to a folder where the data is downloaded for further processing.
59        split: The choice of data split. The dataset does not ship an official split. We use the first
60            20 images for training (of which the last 4 are held out for validation) and the
61            remaining 8 images for testing, following the split convention used in the vessel
62            segmentation literature.
63        annotator: The choice of annotator for the ground-truth vessel maps. There are two independent
64            manual annotations ('1st' and '2nd') per image.
65        download: Whether to download the data if it is not present.
66
67    Returns:
68        List of filepaths for the image data.
69        List of filepaths for the label data.
70    """
71    data_dir = get_chase_db1_data(path=path, download=download)
72
73    assert annotator in ["1st", "2nd"], f"'{annotator}' is not a valid annotator choice."
74
75    image_paths = sorted(glob(os.path.join(data_dir, "Image_*.jpg")))
76    gt_paths = sorted(glob(os.path.join(data_dir, f"Image_*_{annotator}HO.png")))
77
78    assert len(image_paths) == len(gt_paths) and len(image_paths) > 0
79
80    if split == "train":
81        image_paths, gt_paths = image_paths[:16], gt_paths[:16]
82    elif split == "val":
83        image_paths, gt_paths = image_paths[16:20], gt_paths[16:20]
84    elif split == "test":
85        image_paths, gt_paths = image_paths[20:], gt_paths[20:]
86    else:
87        raise ValueError(f"'{split}' is not a valid split.")
88
89    return image_paths, gt_paths

Get paths to the CHASE_DB1 data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The choice of data split. The dataset does not ship an official split. We use the first 20 images for training (of which the last 4 are held out for validation) and the remaining 8 images for testing, following the split convention used in the vessel segmentation literature.
  • annotator: The choice of annotator for the ground-truth vessel maps. There are two independent manual annotations ('1st' and '2nd') per image.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_chase_db1_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], split: Literal['train', 'val', 'test'], annotator: Literal['1st', '2nd'] = '1st', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
 92def get_chase_db1_dataset(
 93    path: Union[os.PathLike, str],
 94    patch_shape: Tuple[int, int],
 95    split: Literal['train', 'val', 'test'],
 96    annotator: Literal['1st', '2nd'] = '1st',
 97    resize_inputs: bool = False,
 98    download: bool = False,
 99    **kwargs
100) -> Dataset:
101    """Get the CHASE_DB1 dataset for segmentation of retinal blood vessels in fundus images.
102
103    Args:
104        path: Filepath to a folder where the data is downloaded for further processing.
105        patch_shape: The patch shape to use for training.
106        split: The choice of data split.
107        annotator: The choice of annotator for the ground-truth vessel maps.
108        resize_inputs: Whether to resize the inputs to the expected patch shape.
109        download: Whether to download the data if it is not present.
110        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
111
112    Returns:
113        The segmentation dataset.
114    """
115    image_paths, gt_paths = get_chase_db1_paths(path=path, split=split, annotator=annotator, download=download)
116
117    if resize_inputs:
118        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
119        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
120            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
121        )
122
123    return torch_em.default_segmentation_dataset(
124        raw_paths=image_paths,
125        raw_key=None,
126        label_paths=gt_paths,
127        label_key=None,
128        patch_shape=patch_shape,
129        is_seg_dataset=False,
130        **kwargs
131    )

Get the CHASE_DB1 dataset for segmentation of retinal blood vessels in fundus images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split.
  • annotator: The choice of annotator for the ground-truth vessel maps.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_chase_db1_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], split: Literal['train', 'val', 'test'], annotator: Literal['1st', '2nd'] = '1st', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
134def get_chase_db1_loader(
135    path: Union[os.PathLike, str],
136    batch_size: int,
137    patch_shape: Tuple[int, int],
138    split: Literal['train', 'val', 'test'],
139    annotator: Literal['1st', '2nd'] = '1st',
140    resize_inputs: bool = False,
141    download: bool = False,
142    **kwargs
143) -> DataLoader:
144    """Get the CHASE_DB1 dataloader for segmentation of retinal blood vessels in fundus images.
145
146    Args:
147        path: Filepath to a folder where the data is downloaded for further processing.
148        batch_size: The batch size for training.
149        patch_shape: The patch shape to use for training.
150        split: The choice of data split.
151        annotator: The choice of annotator for the ground-truth vessel maps.
152        resize_inputs: Whether to resize the inputs to the expected patch shape.
153        download: Whether to download the data if it is not present.
154        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
155
156    Returns:
157        The DataLoader.
158    """
159    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
160    dataset = get_chase_db1_dataset(path, patch_shape, split, annotator, resize_inputs, download, **ds_kwargs)
161    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the CHASE_DB1 dataloader for segmentation of retinal blood vessels in fundus images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split.
  • annotator: The choice of annotator for the ground-truth vessel maps.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.