torch_em.data.datasets.medical.coph100

COph100 is a dataset for fundus image registration from infants, with automatic vessel segmentation masks provided for each retinal fundus image.

The raw fundus images are a subset of the "Retinal Image Dataset of Infants and Retinopathy of Prematurity" (RIDIRP), published in Timkovic et al. - https://doi.org/10.1038/s41597-024-03409-7. COph100 adds manually labeled corresponding point pairs for registration and automatic vessel segmentation masks on top of this subset.

This dataset is from the publication https://doi.org/10.1038/s41597-025-04426-w. Please cite it (and the original RIDIRP publication above) if you use this dataset for your research.

NOTE: The COph100 masks are licensed under CC BY 4.0 (https://creativecommons.org/licenses/by/4.0/), and the underlying RIDIRP fundus images are licensed under CC0 (https://creativecommons.org/publicdomain/zero/1.0/).

  1"""COph100 is a dataset for fundus image registration from infants, with automatic
  2vessel segmentation masks provided for each retinal fundus image.
  3
  4The raw fundus images are a subset of the "Retinal Image Dataset of Infants and
  5Retinopathy of Prematurity" (RIDIRP), published in Timkovic et al. -
  6https://doi.org/10.1038/s41597-024-03409-7. COph100 adds manually labeled corresponding
  7point pairs for registration and automatic vessel segmentation masks on top of this subset.
  8
  9This dataset is from the publication https://doi.org/10.1038/s41597-025-04426-w.
 10Please cite it (and the original RIDIRP publication above) if you use this dataset for your research.
 11
 12NOTE: The COph100 masks are licensed under CC BY 4.0
 13(https://creativecommons.org/licenses/by/4.0/), and the underlying RIDIRP fundus
 14images are licensed under CC0 (https://creativecommons.org/publicdomain/zero/1.0/).
 15"""
 16
 17import os
 18import re
 19import shutil
 20from glob import glob
 21from typing import Union, Tuple, List
 22
 23from torch.utils.data import Dataset, DataLoader
 24
 25import torch_em
 26
 27from .. import util
 28
 29
 30COPH100_URL = "https://ndownloader.figshare.com/files/51235925"
 31COPH100_CHECKSUM = "407ca917280e4f5395b236ffd57096b7ddb2ff3ec362fe150c8ff47c6468e639"
 32
 33RIDIRP_URL = "https://ndownloader.figshare.com/files/43152595"
 34RIDIRP_CHECKSUM = "c07f82340210e8a99bffca99a270a5e6d2c00bb5f1b75e3bc7dac98ad4696163"
 35
 36
 37def _copy_raw_images_from_ridirp(coph100_dir, ridirp_dir):
 38    # The COph100 archive only ships point annotations and vessel masks. The raw fundus
 39    # images have to be copied over from the original RIDIRP images, matched by patient
 40    # id (first three characters of the filename) and examination stage (parsed from the
 41    # "S<stage>" token in the filename), following the logic of the official
 42    # "Copy_COph100_from_ROP.py" script shipped alongside the COph100 archive.
 43    mask_paths = sorted(glob(os.path.join(coph100_dir, "*", "*_mask.png")))
 44    for mask_path in mask_paths:
 45        fname = os.path.basename(mask_path)[:-len("_mask.png")]
 46        patient_id = fname[:3]
 47        stage = int(re.search(r"S(\d+)", fname).group(1))
 48
 49        source_path = os.path.join(ridirp_dir, "images", patient_id, f"{stage:02d}", f"{fname}.jpg")
 50        target_path = os.path.join(os.path.dirname(mask_path), f"{fname}.jpg")
 51
 52        if os.path.exists(target_path):
 53            continue
 54
 55        if not os.path.exists(source_path):
 56            raise RuntimeError(f"Could not find the expected raw image at '{source_path}'.")
 57
 58        shutil.copy2(source_path, target_path)
 59
 60
 61def get_coph100_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 62    """Download the COph100 dataset.
 63
 64    Args:
 65        path: Filepath to a folder where the data is downloaded for further processing.
 66        download: Whether to download the data if it is not present.
 67
 68    Returns:
 69        Filepath where the data is stored.
 70    """
 71    data_dir = os.path.join(path, "COph100")
 72    if os.path.exists(data_dir):
 73        return data_dir
 74
 75    os.makedirs(path, exist_ok=True)
 76
 77    coph100_zip_path = os.path.join(path, "COph100.zip")
 78    util.download_source(path=coph100_zip_path, url=COPH100_URL, download=download, checksum=COPH100_CHECKSUM)
 79    util.unzip(zip_path=coph100_zip_path, dst=data_dir)
 80
 81    ridirp_dir = os.path.join(path, "RIDIRP")
 82    ridirp_zip_path = os.path.join(path, "RIDIRP.zip")
 83    util.download_source(path=ridirp_zip_path, url=RIDIRP_URL, download=download, checksum=RIDIRP_CHECKSUM)
 84    util.unzip(zip_path=ridirp_zip_path, dst=ridirp_dir)
 85
 86    _copy_raw_images_from_ridirp(coph100_dir=data_dir, ridirp_dir=ridirp_dir)
 87
 88    return data_dir
 89
 90
 91def get_coph100_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 92    """Get paths to the COph100 data.
 93
 94    Args:
 95        path: Filepath to a folder where the data is downloaded for further processing.
 96        download: Whether to download the data if it is not present.
 97
 98    Returns:
 99        List of filepaths for the image data.
100        List of filepaths for the label data.
101    """
102    data_dir = get_coph100_data(path=path, download=download)
103
104    gt_paths = sorted(glob(os.path.join(data_dir, "*", "*_mask.png")))
105    image_paths = [p[:-len("_mask.png")] + ".jpg" for p in gt_paths]
106
107    return image_paths, gt_paths
108
109
110def get_coph100_dataset(
111    path: Union[os.PathLike, str],
112    patch_shape: Tuple[int, int],
113    resize_inputs: bool = False,
114    download: bool = False,
115    **kwargs
116) -> Dataset:
117    """Get the COph100 dataset for retinal vessel segmentation in infant fundus images.
118
119    Args:
120        path: Filepath to a folder where the data is downloaded for further processing.
121        patch_shape: The patch shape to use for training.
122        resize_inputs: Whether to resize the inputs to the patch shape.
123        download: Whether to download the data if it is not present.
124        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
125
126    Returns:
127        The segmentation dataset.
128    """
129    image_paths, gt_paths = get_coph100_paths(path, download)
130
131    if resize_inputs:
132        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
133        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
134            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
135        )
136
137    return torch_em.default_segmentation_dataset(
138        raw_paths=image_paths,
139        raw_key=None,
140        label_paths=gt_paths,
141        label_key=None,
142        patch_shape=patch_shape,
143        is_seg_dataset=False,
144        **kwargs
145    )
146
147
148def get_coph100_loader(
149    path: Union[os.PathLike, str],
150    batch_size: int,
151    patch_shape: Tuple[int, int],
152    resize_inputs: bool = False,
153    download: bool = False,
154    **kwargs
155) -> DataLoader:
156    """Get the COph100 dataloader for retinal vessel segmentation in infant fundus images.
157
158    Args:
159        path: Filepath to a folder where the data is downloaded for further processing.
160        batch_size: The batch size for training.
161        patch_shape: The patch shape to use for training.
162        resize_inputs: Whether to resize the inputs to the patch shape.
163        download: Whether to download the data if it is not present.
164        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
165
166    Returns:
167        The DataLoader.
168    """
169    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
170    dataset = get_coph100_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
171    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
COPH100_URL = 'https://ndownloader.figshare.com/files/51235925'
COPH100_CHECKSUM = '407ca917280e4f5395b236ffd57096b7ddb2ff3ec362fe150c8ff47c6468e639'
RIDIRP_URL = 'https://ndownloader.figshare.com/files/43152595'
RIDIRP_CHECKSUM = 'c07f82340210e8a99bffca99a270a5e6d2c00bb5f1b75e3bc7dac98ad4696163'
def get_coph100_data(path: Union[os.PathLike, str], download: bool = False) -> str:
62def get_coph100_data(path: Union[os.PathLike, str], download: bool = False) -> str:
63    """Download the COph100 dataset.
64
65    Args:
66        path: Filepath to a folder where the data is downloaded for further processing.
67        download: Whether to download the data if it is not present.
68
69    Returns:
70        Filepath where the data is stored.
71    """
72    data_dir = os.path.join(path, "COph100")
73    if os.path.exists(data_dir):
74        return data_dir
75
76    os.makedirs(path, exist_ok=True)
77
78    coph100_zip_path = os.path.join(path, "COph100.zip")
79    util.download_source(path=coph100_zip_path, url=COPH100_URL, download=download, checksum=COPH100_CHECKSUM)
80    util.unzip(zip_path=coph100_zip_path, dst=data_dir)
81
82    ridirp_dir = os.path.join(path, "RIDIRP")
83    ridirp_zip_path = os.path.join(path, "RIDIRP.zip")
84    util.download_source(path=ridirp_zip_path, url=RIDIRP_URL, download=download, checksum=RIDIRP_CHECKSUM)
85    util.unzip(zip_path=ridirp_zip_path, dst=ridirp_dir)
86
87    _copy_raw_images_from_ridirp(coph100_dir=data_dir, ridirp_dir=ridirp_dir)
88
89    return data_dir

Download the COph100 dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is stored.

def get_coph100_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 92def get_coph100_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 93    """Get paths to the COph100 data.
 94
 95    Args:
 96        path: Filepath to a folder where the data is downloaded for further processing.
 97        download: Whether to download the data if it is not present.
 98
 99    Returns:
100        List of filepaths for the image data.
101        List of filepaths for the label data.
102    """
103    data_dir = get_coph100_data(path=path, download=download)
104
105    gt_paths = sorted(glob(os.path.join(data_dir, "*", "*_mask.png")))
106    image_paths = [p[:-len("_mask.png")] + ".jpg" for p in gt_paths]
107
108    return image_paths, gt_paths

Get paths to the COph100 data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_coph100_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
111def get_coph100_dataset(
112    path: Union[os.PathLike, str],
113    patch_shape: Tuple[int, int],
114    resize_inputs: bool = False,
115    download: bool = False,
116    **kwargs
117) -> Dataset:
118    """Get the COph100 dataset for retinal vessel segmentation in infant fundus images.
119
120    Args:
121        path: Filepath to a folder where the data is downloaded for further processing.
122        patch_shape: The patch shape to use for training.
123        resize_inputs: Whether to resize the inputs to the patch shape.
124        download: Whether to download the data if it is not present.
125        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
126
127    Returns:
128        The segmentation dataset.
129    """
130    image_paths, gt_paths = get_coph100_paths(path, download)
131
132    if resize_inputs:
133        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
134        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
135            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
136        )
137
138    return torch_em.default_segmentation_dataset(
139        raw_paths=image_paths,
140        raw_key=None,
141        label_paths=gt_paths,
142        label_key=None,
143        patch_shape=patch_shape,
144        is_seg_dataset=False,
145        **kwargs
146    )

Get the COph100 dataset for retinal vessel segmentation in infant fundus images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_coph100_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
149def get_coph100_loader(
150    path: Union[os.PathLike, str],
151    batch_size: int,
152    patch_shape: Tuple[int, int],
153    resize_inputs: bool = False,
154    download: bool = False,
155    **kwargs
156) -> DataLoader:
157    """Get the COph100 dataloader for retinal vessel segmentation in infant fundus images.
158
159    Args:
160        path: Filepath to a folder where the data is downloaded for further processing.
161        batch_size: The batch size for training.
162        patch_shape: The patch shape to use for training.
163        resize_inputs: Whether to resize the inputs to the patch shape.
164        download: Whether to download the data if it is not present.
165        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
166
167    Returns:
168        The DataLoader.
169    """
170    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
171    dataset = get_coph100_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
172    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the COph100 dataloader for retinal vessel segmentation in infant fundus images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.