torch_em.data.datasets.medical.afio

The AFIO dataset contains annotations for retinal vessel, artery and vein segmentation in fundus images.

The dataset consists of 100 colour fundus images (86 macula-centred and 14 optic disc-centred) acquired at the Armed Forces Institute of Ophthalmology (AFIO), Rawalpindi, Pakistan, and manually annotated by four expert ophthalmologists. The publicly downloadable archive ships pixel-level annotations for the retinal vessel network and for the separated artery / vein networks (plus a combined "both" overlay). The optic nerve head, hard exudate and cotton-wool spot annotations that are mentioned in the associated publication are not part of the downloadable archive.

The dataset is located at https://data.mendeley.com/datasets/3csr652p9y/2 (CC BY 4.0). This dataset is from the publication https://doi.org/10.1016/j.dib.2020.105282. Please cite it if you use this dataset for your research.

  1"""The AFIO dataset contains annotations for retinal vessel, artery and vein segmentation
  2in fundus images.
  3
  4The dataset consists of 100 colour fundus images (86 macula-centred and 14 optic disc-centred)
  5acquired at the Armed Forces Institute of Ophthalmology (AFIO), Rawalpindi, Pakistan, and manually
  6annotated by four expert ophthalmologists. The publicly downloadable archive ships pixel-level
  7annotations for the retinal vessel network and for the separated artery / vein networks (plus a
  8combined "both" overlay). The optic nerve head, hard exudate and cotton-wool spot annotations that
  9are mentioned in the associated publication are not part of the downloadable archive.
 10
 11The dataset is located at https://data.mendeley.com/datasets/3csr652p9y/2 (CC BY 4.0).
 12This dataset is from the publication https://doi.org/10.1016/j.dib.2020.105282.
 13Please cite it if you use this dataset for your research.
 14"""
 15
 16import os
 17from glob import glob
 18from tqdm import tqdm
 19from pathlib import Path
 20from natsort import natsorted
 21from typing import Union, Literal, Tuple, List
 22
 23import numpy as np
 24import imageio.v3 as imageio
 25
 26from torch.utils.data import Dataset, DataLoader
 27
 28import torch_em
 29
 30from .. import util
 31
 32
 33URL = "https://data.mendeley.com/public-files/datasets/3csr652p9y/files/5c07e45a-5f3f-407b-8bdb-16332a84fa23/file_downloaded"  # noqa
 34CHECKSUM = "f0af3cc8714e1eaff5d2b5a3e0b77f8c6166a0d18322dd2685c2c6ed325fc230"
 35
 36TASKS = ["vessels", "arteries", "veins", "both"]
 37
 38
 39def get_afio_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 40    """Download the AFIO dataset.
 41
 42    Args:
 43        path: Filepath to a folder where the data is downloaded for further processing.
 44        download: Whether to download the data if it is not present.
 45
 46    Returns:
 47        Filepath where the data is downloaded.
 48    """
 49    data_dir = os.path.join(path, "AV")
 50    if os.path.exists(data_dir):
 51        return data_dir
 52
 53    os.makedirs(path, exist_ok=True)
 54
 55    zip_path = os.path.join(path, "AV.zip")
 56    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 57    util.unzip(zip_path=zip_path, dst=path)
 58
 59    return data_dir
 60
 61
 62def _match_annotation(annotation_paths: List[str], task: str) -> str:
 63    # The annotations were exported from Illustrator by hand, so the suffixes have several typos,
 64    # e.g. 'arteries' / 'artery' / 'artry' / 'atertries' and 'veins' / 'vein' / 'veinds' / 'veisn'.
 65    # None of the artery variants contain the letter 'v', so this is used to disambiguate them
 66    # from the vein variants once the unambiguous 'vessels' and 'both' / 'map' suffixes are removed.
 67    vessel_paths = [p for p in annotation_paths if "vessel" in Path(p).stem.lower()]
 68    both_paths = [p for p in annotation_paths if "both" in Path(p).stem.lower() or "map" in Path(p).stem.lower()]
 69    remaining_paths = [p for p in annotation_paths if p not in vessel_paths and p not in both_paths]
 70    vein_paths = [p for p in remaining_paths if "v" in Path(p).stem.lower().split("--")[-1]]
 71    artery_paths = [p for p in remaining_paths if p not in vein_paths]
 72
 73    task_to_paths = {"vessels": vessel_paths, "both": both_paths, "veins": vein_paths, "arteries": artery_paths}
 74    matches = task_to_paths[task]
 75    assert len(matches) == 1, f"Expected exactly one '{task}' annotation, found {matches}."
 76    return matches[0]
 77
 78
 79def get_afio_paths(
 80    path: Union[os.PathLike, str],
 81    task: Literal["vessels", "arteries", "veins", "both"] = "vessels",
 82    download: bool = False,
 83) -> Tuple[List[str], List[str]]:
 84    """Get paths to the AFIO data.
 85
 86    Args:
 87        path: Filepath to a folder where the data is downloaded for further processing.
 88        task: The choice of annotation. One of 'vessels', 'arteries', 'veins' or 'both'.
 89        download: Whether to download the data if it is not present.
 90
 91    Returns:
 92        List of filepaths for the image data.
 93        List of filepaths for the label data.
 94    """
 95    data_dir = get_afio_data(path=path, download=download)
 96
 97    assert task in TASKS, f"'{task}' is not a valid task. Please choose from {TASKS}."
 98
 99    image_dirs = natsorted(glob(os.path.join(data_dir, "IM*")))
100    assert len(image_dirs) == 100, f"Expected 100 image folders, found {len(image_dirs)}."
101
102    neu_gt_dir = os.path.join(data_dir, "preprocessed", task)
103    os.makedirs(neu_gt_dir, exist_ok=True)
104
105    image_paths, gt_paths = [], []
106    for image_dir in tqdm(image_dirs, desc=f"Preprocessing '{task}' labels"):
107        name = os.path.basename(image_dir)
108        # A couple of image folders have an extra (redundant) nesting level, e.g.
109        # 'AV/IM000189/IM000189/IM000189.JPG' instead of 'AV/IM000189/IM000189.JPG',
110        # so the raw image and annotations are searched for recursively.
111        image_matches = glob(os.path.join(image_dir, "**", f"{name}.JPG"), recursive=True)
112        assert len(image_matches) == 1, f"Expected exactly one raw image for '{name}', found {image_matches}."
113        image_path = image_matches[0]
114
115        annotation_paths = natsorted(glob(os.path.join(image_dir, "**", f"{name}--*.jpg"), recursive=True))
116        raw_gt_path = _match_annotation(annotation_paths, task)
117
118        gt_path = os.path.join(neu_gt_dir, f"{name}.tif")
119        if not os.path.exists(gt_path):
120            # The masks are lightly JPEG-compressed overlays with a bright background and a dark
121            # foreground structure (vessel / artery / vein / combined network), so they are
122            # binarized into a uint8 (0, 1) label map with the foreground being the darker pixels.
123            raw_gt = imageio.imread(raw_gt_path)
124            gray_gt = raw_gt.mean(axis=-1) if raw_gt.ndim == 3 else raw_gt
125            binary_gt = (gray_gt < 128).astype(np.uint8)
126            imageio.imwrite(gt_path, binary_gt)
127
128        image_paths.append(image_path)
129        gt_paths.append(gt_path)
130
131    return image_paths, gt_paths
132
133
134def get_afio_dataset(
135    path: Union[os.PathLike, str],
136    patch_shape: Tuple[int, int],
137    task: Literal["vessels", "arteries", "veins", "both"] = "vessels",
138    resize_inputs: bool = False,
139    download: bool = False,
140    **kwargs
141) -> Dataset:
142    """Get the AFIO dataset for segmentation of retinal vessels, arteries and veins in fundus images.
143
144    Args:
145        path: Filepath to a folder where the data is downloaded for further processing.
146        patch_shape: The patch shape to use for training.
147        task: The choice of annotation. One of 'vessels', 'arteries', 'veins' or 'both'.
148        resize_inputs: Whether to resize the inputs to the expected patch shape.
149        download: Whether to download the data if it is not present.
150        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
151
152    Returns:
153        The segmentation dataset.
154    """
155    image_paths, gt_paths = get_afio_paths(path, task, download)
156
157    if resize_inputs:
158        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
159        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
160            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
161        )
162
163    return torch_em.default_segmentation_dataset(
164        raw_paths=image_paths,
165        raw_key=None,
166        label_paths=gt_paths,
167        label_key=None,
168        is_seg_dataset=False,
169        patch_shape=patch_shape,
170        **kwargs
171    )
172
173
174def get_afio_loader(
175    path: Union[os.PathLike, str],
176    batch_size: int,
177    patch_shape: Tuple[int, int],
178    task: Literal["vessels", "arteries", "veins", "both"] = "vessels",
179    resize_inputs: bool = False,
180    download: bool = False,
181    **kwargs
182) -> DataLoader:
183    """Get the AFIO dataloader for segmentation of retinal vessels, arteries and veins in fundus images.
184
185    Args:
186        path: Filepath to a folder where the data is downloaded for further processing.
187        batch_size: The batch size for training.
188        patch_shape: The patch shape to use for training.
189        task: The choice of annotation. One of 'vessels', 'arteries', 'veins' or 'both'.
190        resize_inputs: Whether to resize the inputs to the expected patch shape.
191        download: Whether to download the data if it is not present.
192        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
193
194    Returns:
195        The DataLoader.
196    """
197    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
198    dataset = get_afio_dataset(path, patch_shape, task, resize_inputs, download, **ds_kwargs)
199    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://data.mendeley.com/public-files/datasets/3csr652p9y/files/5c07e45a-5f3f-407b-8bdb-16332a84fa23/file_downloaded'
CHECKSUM = 'f0af3cc8714e1eaff5d2b5a3e0b77f8c6166a0d18322dd2685c2c6ed325fc230'
TASKS = ['vessels', 'arteries', 'veins', 'both']
def get_afio_data(path: Union[os.PathLike, str], download: bool = False) -> str:
40def get_afio_data(path: Union[os.PathLike, str], download: bool = False) -> str:
41    """Download the AFIO dataset.
42
43    Args:
44        path: Filepath to a folder where the data is downloaded for further processing.
45        download: Whether to download the data if it is not present.
46
47    Returns:
48        Filepath where the data is downloaded.
49    """
50    data_dir = os.path.join(path, "AV")
51    if os.path.exists(data_dir):
52        return data_dir
53
54    os.makedirs(path, exist_ok=True)
55
56    zip_path = os.path.join(path, "AV.zip")
57    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
58    util.unzip(zip_path=zip_path, dst=path)
59
60    return data_dir

Download the AFIO dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_afio_paths( path: Union[os.PathLike, str], task: Literal['vessels', 'arteries', 'veins', 'both'] = 'vessels', download: bool = False) -> Tuple[List[str], List[str]]:
 80def get_afio_paths(
 81    path: Union[os.PathLike, str],
 82    task: Literal["vessels", "arteries", "veins", "both"] = "vessels",
 83    download: bool = False,
 84) -> Tuple[List[str], List[str]]:
 85    """Get paths to the AFIO data.
 86
 87    Args:
 88        path: Filepath to a folder where the data is downloaded for further processing.
 89        task: The choice of annotation. One of 'vessels', 'arteries', 'veins' or 'both'.
 90        download: Whether to download the data if it is not present.
 91
 92    Returns:
 93        List of filepaths for the image data.
 94        List of filepaths for the label data.
 95    """
 96    data_dir = get_afio_data(path=path, download=download)
 97
 98    assert task in TASKS, f"'{task}' is not a valid task. Please choose from {TASKS}."
 99
100    image_dirs = natsorted(glob(os.path.join(data_dir, "IM*")))
101    assert len(image_dirs) == 100, f"Expected 100 image folders, found {len(image_dirs)}."
102
103    neu_gt_dir = os.path.join(data_dir, "preprocessed", task)
104    os.makedirs(neu_gt_dir, exist_ok=True)
105
106    image_paths, gt_paths = [], []
107    for image_dir in tqdm(image_dirs, desc=f"Preprocessing '{task}' labels"):
108        name = os.path.basename(image_dir)
109        # A couple of image folders have an extra (redundant) nesting level, e.g.
110        # 'AV/IM000189/IM000189/IM000189.JPG' instead of 'AV/IM000189/IM000189.JPG',
111        # so the raw image and annotations are searched for recursively.
112        image_matches = glob(os.path.join(image_dir, "**", f"{name}.JPG"), recursive=True)
113        assert len(image_matches) == 1, f"Expected exactly one raw image for '{name}', found {image_matches}."
114        image_path = image_matches[0]
115
116        annotation_paths = natsorted(glob(os.path.join(image_dir, "**", f"{name}--*.jpg"), recursive=True))
117        raw_gt_path = _match_annotation(annotation_paths, task)
118
119        gt_path = os.path.join(neu_gt_dir, f"{name}.tif")
120        if not os.path.exists(gt_path):
121            # The masks are lightly JPEG-compressed overlays with a bright background and a dark
122            # foreground structure (vessel / artery / vein / combined network), so they are
123            # binarized into a uint8 (0, 1) label map with the foreground being the darker pixels.
124            raw_gt = imageio.imread(raw_gt_path)
125            gray_gt = raw_gt.mean(axis=-1) if raw_gt.ndim == 3 else raw_gt
126            binary_gt = (gray_gt < 128).astype(np.uint8)
127            imageio.imwrite(gt_path, binary_gt)
128
129        image_paths.append(image_path)
130        gt_paths.append(gt_path)
131
132    return image_paths, gt_paths

Get paths to the AFIO data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • task: The choice of annotation. One of 'vessels', 'arteries', 'veins' or 'both'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_afio_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], task: Literal['vessels', 'arteries', 'veins', 'both'] = 'vessels', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
135def get_afio_dataset(
136    path: Union[os.PathLike, str],
137    patch_shape: Tuple[int, int],
138    task: Literal["vessels", "arteries", "veins", "both"] = "vessels",
139    resize_inputs: bool = False,
140    download: bool = False,
141    **kwargs
142) -> Dataset:
143    """Get the AFIO dataset for segmentation of retinal vessels, arteries and veins in fundus images.
144
145    Args:
146        path: Filepath to a folder where the data is downloaded for further processing.
147        patch_shape: The patch shape to use for training.
148        task: The choice of annotation. One of 'vessels', 'arteries', 'veins' or 'both'.
149        resize_inputs: Whether to resize the inputs to the expected patch shape.
150        download: Whether to download the data if it is not present.
151        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
152
153    Returns:
154        The segmentation dataset.
155    """
156    image_paths, gt_paths = get_afio_paths(path, task, download)
157
158    if resize_inputs:
159        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
160        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
161            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
162        )
163
164    return torch_em.default_segmentation_dataset(
165        raw_paths=image_paths,
166        raw_key=None,
167        label_paths=gt_paths,
168        label_key=None,
169        is_seg_dataset=False,
170        patch_shape=patch_shape,
171        **kwargs
172    )

Get the AFIO dataset for segmentation of retinal vessels, arteries and veins in fundus images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • task: The choice of annotation. One of 'vessels', 'arteries', 'veins' or 'both'.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_afio_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], task: Literal['vessels', 'arteries', 'veins', 'both'] = 'vessels', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
175def get_afio_loader(
176    path: Union[os.PathLike, str],
177    batch_size: int,
178    patch_shape: Tuple[int, int],
179    task: Literal["vessels", "arteries", "veins", "both"] = "vessels",
180    resize_inputs: bool = False,
181    download: bool = False,
182    **kwargs
183) -> DataLoader:
184    """Get the AFIO dataloader for segmentation of retinal vessels, arteries and veins in fundus images.
185
186    Args:
187        path: Filepath to a folder where the data is downloaded for further processing.
188        batch_size: The batch size for training.
189        patch_shape: The patch shape to use for training.
190        task: The choice of annotation. One of 'vessels', 'arteries', 'veins' or 'both'.
191        resize_inputs: Whether to resize the inputs to the expected patch shape.
192        download: Whether to download the data if it is not present.
193        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
194
195    Returns:
196        The DataLoader.
197    """
198    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
199    dataset = get_afio_dataset(path, patch_shape, task, resize_inputs, download, **ds_kwargs)
200    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the AFIO dataloader for segmentation of retinal vessels, arteries and veins in fundus images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • task: The choice of annotation. One of 'vessels', 'arteries', 'veins' or 'both'.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.