torch_em.data.datasets.medical.stare

The STARE dataset contains annotations for retinal vessel segmentation in fundus images.

STARE (STructured Analysis of the Retina) is located at http://cecas.clemson.edu/~ahoover/stare/. It provides 20 fundus images along with two independent sets of hand-labeled vessel maps, created by Adam Hoover ('ah') and Valentina Kouznetsova ('vk').

The dataset is from the publication https://doi.org/10.1109/42.845178. Please cite it if you use this dataset for your research.

  1"""The STARE dataset contains annotations for retinal vessel segmentation in fundus images.
  2
  3STARE (STructured Analysis of the Retina) is located at http://cecas.clemson.edu/~ahoover/stare/.
  4It provides 20 fundus images along with two independent sets of hand-labeled vessel maps, created
  5by Adam Hoover ('ah') and Valentina Kouznetsova ('vk').
  6
  7The dataset is from the publication https://doi.org/10.1109/42.845178.
  8Please cite it if you use this dataset for your research.
  9"""
 10
 11import os
 12import gzip
 13import shutil
 14from glob import glob
 15from typing import Union, Tuple, Literal, List
 16
 17from torch.utils.data import Dataset, DataLoader
 18
 19import torch_em
 20
 21from .. import util
 22
 23
 24URL = {
 25    "images": "http://cecas.clemson.edu/~ahoover/stare/probing/stare-images.tar",
 26    "ah": "http://cecas.clemson.edu/~ahoover/stare/probing/labels-ah.tar",
 27    "vk": "http://cecas.clemson.edu/~ahoover/stare/probing/labels-vk.tar",
 28}
 29
 30CHECKSUM = {
 31    "images": "5f7b509b6067cad4f1be84933145d783de9ec087b3eaaf4db0103dd0144dd433",
 32    "ah": "ebf2f1e17ca955f24579d9edd990e2dae79a5c82def69f0985d8e24f826ddd2f",
 33    "vk": "47474a701536b0cfdb369fdce012be36141e9f44d80387f0179446b5cb0f5576",
 34}
 35
 36
 37def _unpack_gz_files(dir_path):
 38    gz_paths = sorted(glob(os.path.join(dir_path, "*.gz")))
 39    for gz_path in gz_paths:
 40        out_path = gz_path[:-len(".gz")]
 41        if os.path.exists(out_path):
 42            continue
 43        with gzip.open(gz_path, "rb") as f_in, open(out_path, "wb") as f_out:
 44            shutil.copyfileobj(f_in, f_out)
 45
 46
 47def get_stare_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 48    """Download the STARE dataset.
 49
 50    Args:
 51        path: Filepath to a folder where the data is downloaded for further processing.
 52        download: Whether to download the data if it is not present.
 53
 54    Returns:
 55        Filepath where the data is downloaded.
 56    """
 57    images_dir = os.path.join(path, "images")
 58    if os.path.exists(images_dir):
 59        return path
 60
 61    os.makedirs(path, exist_ok=True)
 62
 63    for name, target_dir in [("images", "images"), ("ah", "labels-ah"), ("vk", "labels-vk")]:
 64        tar_path = os.path.join(path, f"{name}.tar")
 65        util.download_source(path=tar_path, url=URL[name], download=download, checksum=CHECKSUM[name])
 66
 67        extract_dir = os.path.join(path, target_dir)
 68        os.makedirs(extract_dir, exist_ok=True)
 69        shutil.unpack_archive(tar_path, extract_dir, format="tar")
 70
 71        _unpack_gz_files(extract_dir)
 72
 73    return path
 74
 75
 76def get_stare_paths(
 77    path: Union[os.PathLike, str],
 78    annotator: Literal["ah", "vk"] = "ah",
 79    download: bool = False,
 80) -> Tuple[List[str], List[str]]:
 81    """Get paths to the STARE data.
 82
 83    Args:
 84        path: Filepath to a folder where the data is downloaded for further processing.
 85        annotator: The choice of annotator for the ground-truth vessel maps. There are two independent
 86            manual annotations, provided by 'ah' (Adam Hoover) and 'vk' (Valentina Kouznetsova).
 87        download: Whether to download the data if it is not present.
 88
 89    Returns:
 90        List of filepaths for the image data.
 91        List of filepaths for the label data.
 92    """
 93    data_dir = get_stare_data(path=path, download=download)
 94
 95    assert annotator in ["ah", "vk"], f"'{annotator}' is not a valid annotator choice."
 96
 97    image_paths = sorted(glob(os.path.join(data_dir, "images", "*.ppm")))
 98    gt_paths = sorted(glob(os.path.join(data_dir, f"labels-{annotator}", f"*.{annotator}.ppm")))
 99
100    assert len(image_paths) == len(gt_paths) and len(image_paths) > 0
101
102    return image_paths, gt_paths
103
104
105def get_stare_dataset(
106    path: Union[os.PathLike, str],
107    patch_shape: Tuple[int, int],
108    annotator: Literal["ah", "vk"] = "ah",
109    resize_inputs: bool = False,
110    download: bool = False,
111    **kwargs
112) -> Dataset:
113    """Get the STARE dataset for segmentation of retinal blood vessels in fundus images.
114
115    Args:
116        path: Filepath to a folder where the data is downloaded for further processing.
117        patch_shape: The patch shape to use for training.
118        annotator: The choice of annotator for the ground-truth vessel maps.
119        resize_inputs: Whether to resize the inputs to the expected patch shape.
120        download: Whether to download the data if it is not present.
121        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
122
123    Returns:
124        The segmentation dataset.
125    """
126    image_paths, gt_paths = get_stare_paths(path=path, annotator=annotator, download=download)
127
128    if resize_inputs:
129        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
130        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
131            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
132        )
133
134    return torch_em.default_segmentation_dataset(
135        raw_paths=image_paths,
136        raw_key=None,
137        label_paths=gt_paths,
138        label_key=None,
139        patch_shape=patch_shape,
140        is_seg_dataset=False,
141        **kwargs
142    )
143
144
145def get_stare_loader(
146    path: Union[os.PathLike, str],
147    batch_size: int,
148    patch_shape: Tuple[int, int],
149    annotator: Literal["ah", "vk"] = "ah",
150    resize_inputs: bool = False,
151    download: bool = False,
152    **kwargs
153) -> DataLoader:
154    """Get the STARE dataloader for segmentation of retinal blood vessels in fundus images.
155
156    Args:
157        path: Filepath to a folder where the data is downloaded for further processing.
158        batch_size: The batch size for training.
159        patch_shape: The patch shape to use for training.
160        annotator: The choice of annotator for the ground-truth vessel maps.
161        resize_inputs: Whether to resize the inputs to the expected patch shape.
162        download: Whether to download the data if it is not present.
163        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
164
165    Returns:
166        The DataLoader.
167    """
168    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
169    dataset = get_stare_dataset(path, patch_shape, annotator, resize_inputs, download, **ds_kwargs)
170    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = {'images': 'http://cecas.clemson.edu/~ahoover/stare/probing/stare-images.tar', 'ah': 'http://cecas.clemson.edu/~ahoover/stare/probing/labels-ah.tar', 'vk': 'http://cecas.clemson.edu/~ahoover/stare/probing/labels-vk.tar'}
CHECKSUM = {'images': '5f7b509b6067cad4f1be84933145d783de9ec087b3eaaf4db0103dd0144dd433', 'ah': 'ebf2f1e17ca955f24579d9edd990e2dae79a5c82def69f0985d8e24f826ddd2f', 'vk': '47474a701536b0cfdb369fdce012be36141e9f44d80387f0179446b5cb0f5576'}
def get_stare_data(path: Union[os.PathLike, str], download: bool = False) -> str:
48def get_stare_data(path: Union[os.PathLike, str], download: bool = False) -> str:
49    """Download the STARE dataset.
50
51    Args:
52        path: Filepath to a folder where the data is downloaded for further processing.
53        download: Whether to download the data if it is not present.
54
55    Returns:
56        Filepath where the data is downloaded.
57    """
58    images_dir = os.path.join(path, "images")
59    if os.path.exists(images_dir):
60        return path
61
62    os.makedirs(path, exist_ok=True)
63
64    for name, target_dir in [("images", "images"), ("ah", "labels-ah"), ("vk", "labels-vk")]:
65        tar_path = os.path.join(path, f"{name}.tar")
66        util.download_source(path=tar_path, url=URL[name], download=download, checksum=CHECKSUM[name])
67
68        extract_dir = os.path.join(path, target_dir)
69        os.makedirs(extract_dir, exist_ok=True)
70        shutil.unpack_archive(tar_path, extract_dir, format="tar")
71
72        _unpack_gz_files(extract_dir)
73
74    return path

Download the STARE dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_stare_paths( path: Union[os.PathLike, str], annotator: Literal['ah', 'vk'] = 'ah', download: bool = False) -> Tuple[List[str], List[str]]:
 77def get_stare_paths(
 78    path: Union[os.PathLike, str],
 79    annotator: Literal["ah", "vk"] = "ah",
 80    download: bool = False,
 81) -> Tuple[List[str], List[str]]:
 82    """Get paths to the STARE data.
 83
 84    Args:
 85        path: Filepath to a folder where the data is downloaded for further processing.
 86        annotator: The choice of annotator for the ground-truth vessel maps. There are two independent
 87            manual annotations, provided by 'ah' (Adam Hoover) and 'vk' (Valentina Kouznetsova).
 88        download: Whether to download the data if it is not present.
 89
 90    Returns:
 91        List of filepaths for the image data.
 92        List of filepaths for the label data.
 93    """
 94    data_dir = get_stare_data(path=path, download=download)
 95
 96    assert annotator in ["ah", "vk"], f"'{annotator}' is not a valid annotator choice."
 97
 98    image_paths = sorted(glob(os.path.join(data_dir, "images", "*.ppm")))
 99    gt_paths = sorted(glob(os.path.join(data_dir, f"labels-{annotator}", f"*.{annotator}.ppm")))
100
101    assert len(image_paths) == len(gt_paths) and len(image_paths) > 0
102
103    return image_paths, gt_paths

Get paths to the STARE data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • annotator: The choice of annotator for the ground-truth vessel maps. There are two independent manual annotations, provided by 'ah' (Adam Hoover) and 'vk' (Valentina Kouznetsova).
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_stare_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], annotator: Literal['ah', 'vk'] = 'ah', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
106def get_stare_dataset(
107    path: Union[os.PathLike, str],
108    patch_shape: Tuple[int, int],
109    annotator: Literal["ah", "vk"] = "ah",
110    resize_inputs: bool = False,
111    download: bool = False,
112    **kwargs
113) -> Dataset:
114    """Get the STARE dataset for segmentation of retinal blood vessels in fundus images.
115
116    Args:
117        path: Filepath to a folder where the data is downloaded for further processing.
118        patch_shape: The patch shape to use for training.
119        annotator: The choice of annotator for the ground-truth vessel maps.
120        resize_inputs: Whether to resize the inputs to the expected patch shape.
121        download: Whether to download the data if it is not present.
122        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
123
124    Returns:
125        The segmentation dataset.
126    """
127    image_paths, gt_paths = get_stare_paths(path=path, annotator=annotator, download=download)
128
129    if resize_inputs:
130        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
131        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
132            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
133        )
134
135    return torch_em.default_segmentation_dataset(
136        raw_paths=image_paths,
137        raw_key=None,
138        label_paths=gt_paths,
139        label_key=None,
140        patch_shape=patch_shape,
141        is_seg_dataset=False,
142        **kwargs
143    )

Get the STARE dataset for segmentation of retinal blood vessels in fundus images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • annotator: The choice of annotator for the ground-truth vessel maps.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_stare_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], annotator: Literal['ah', 'vk'] = 'ah', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
146def get_stare_loader(
147    path: Union[os.PathLike, str],
148    batch_size: int,
149    patch_shape: Tuple[int, int],
150    annotator: Literal["ah", "vk"] = "ah",
151    resize_inputs: bool = False,
152    download: bool = False,
153    **kwargs
154) -> DataLoader:
155    """Get the STARE dataloader for segmentation of retinal blood vessels in fundus images.
156
157    Args:
158        path: Filepath to a folder where the data is downloaded for further processing.
159        batch_size: The batch size for training.
160        patch_shape: The patch shape to use for training.
161        annotator: The choice of annotator for the ground-truth vessel maps.
162        resize_inputs: Whether to resize the inputs to the expected patch shape.
163        download: Whether to download the data if it is not present.
164        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
165
166    Returns:
167        The DataLoader.
168    """
169    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
170    dataset = get_stare_dataset(path, patch_shape, annotator, resize_inputs, download, **ds_kwargs)
171    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the STARE dataloader for segmentation of retinal blood vessels in fundus images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • annotator: The choice of annotator for the ground-truth vessel maps.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.