torch_em.data.datasets.medical.ocutox

The OCUTOX dataset contains annotations for active and inactive lesion segmentation in fundus images of patients with ocular toxoplasmosis.

The dataset contains 412 fundus images collected at the Hospital de Clinicas and the Hospital General Pediatrico Acosta Nu medical centers in Asuncion, Paraguay, of which 280 images (with active and / or inactive toxoplasmosis lesions) have pixel-level lesion masks delineated by ophthalmologists. The remaining images are labeled 'healthy' and do not have a lesion mask.

This dataset is located at https://doi.org/10.5281/zenodo.5156940 (CC BY 4.0). Please cite it if you use this dataset for your research.

  1"""The OCUTOX dataset contains annotations for active and inactive lesion segmentation
  2in fundus images of patients with ocular toxoplasmosis.
  3
  4The dataset contains 412 fundus images collected at the Hospital de Clinicas and the
  5Hospital General Pediatrico Acosta Nu medical centers in Asuncion, Paraguay, of which
  6280 images (with active and / or inactive toxoplasmosis lesions) have pixel-level lesion
  7masks delineated by ophthalmologists. The remaining images are labeled 'healthy' and do
  8not have a lesion mask.
  9
 10This dataset is located at https://doi.org/10.5281/zenodo.5156940 (CC BY 4.0).
 11Please cite it if you use this dataset for your research.
 12"""
 13
 14import os
 15import re
 16from glob import glob
 17from typing import Union, Tuple, List
 18
 19from torch.utils.data import Dataset, DataLoader
 20
 21import torch_em
 22
 23from .. import util
 24
 25
 26URL = "https://zenodo.org/records/5156940/files/Ocular_Toxoplasmosis_Data_V3.zip"
 27CHECKSUM = "d088dcfe6c678923eed20e18052935cff70a3e0ede8f87f06406caa734727ece"
 28
 29# Masks carry suffixes for lesion sub-regions of the same image, eg. '-a' (active lesion),
 30# '-i' (inactive lesion) and numeric variants ('-2', '-3', '-a-2', ...).
 31MASK_SUFFIX_PATTERN = re.compile(r"-(?:a|i)(?:-\d+)?$|-\d+$", flags=re.IGNORECASE)
 32
 33
 34def get_ocutox_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 35    """Download the OCUTOX dataset.
 36
 37    Args:
 38        path: Filepath to a folder where the data is downloaded for further processing.
 39        download: Whether to download the data if it is not present.
 40
 41    Returns:
 42        Filepath where the data is downloaded.
 43    """
 44    data_dir = os.path.join(path, "images")
 45    if os.path.exists(data_dir):
 46        return path
 47
 48    os.makedirs(path, exist_ok=True)
 49
 50    zip_path = os.path.join(path, "Ocular_Toxoplasmosis_Data_V3.zip")
 51    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 52    util.unzip(zip_path=zip_path, dst=path)
 53
 54    return path
 55
 56
 57def get_ocutox_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 58    """Get paths to the OCUTOX data.
 59
 60    Args:
 61        path: Filepath to a folder where the data is downloaded for further processing.
 62        download: Whether to download the data if it is not present.
 63
 64    Returns:
 65        List of filepaths for the image data.
 66        List of filepaths for the label data.
 67    """
 68    data_dir = get_ocutox_data(path=path, download=download)
 69
 70    image_dir = os.path.join(data_dir, "images")
 71    mask_dir = os.path.join(data_dir, "masks")
 72
 73    image_files = {os.path.basename(p).lower(): p for p in glob(os.path.join(image_dir, "*.*"))}
 74    mask_paths = sorted(glob(os.path.join(mask_dir, "*.*")))
 75
 76    image_paths, matched_mask_paths = [], []
 77    for mask_path in mask_paths:
 78        fname = os.path.basename(mask_path)
 79        stem, ext = os.path.splitext(fname.lower())
 80        base_stem = MASK_SUFFIX_PATTERN.sub("", stem)
 81        image_path = image_files.get(base_stem + ext)
 82        if image_path is None:
 83            raise RuntimeError(f"Could not find the matching image for the mask at '{mask_path}'.")
 84
 85        image_paths.append(image_path)
 86        matched_mask_paths.append(mask_path)
 87
 88    if len(image_paths) == 0 or len(image_paths) != len(matched_mask_paths):
 89        raise RuntimeError("Something went wrong with fetching the image and label paths.")
 90
 91    return image_paths, matched_mask_paths
 92
 93
 94def get_ocutox_dataset(
 95    path: Union[os.PathLike, str],
 96    patch_shape: Tuple[int, int],
 97    resize_inputs: bool = False,
 98    download: bool = False,
 99    **kwargs
100) -> Dataset:
101    """Get the OCUTOX dataset for segmentation of ocular toxoplasmosis lesions in fundus images.
102
103    Args:
104        path: Filepath to a folder where the data is downloaded for further processing.
105        patch_shape: The patch shape to use for training.
106        resize_inputs: Whether to resize the inputs to the expected patch shape.
107        download: Whether to download the data if it is not present.
108        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
109
110    Returns:
111        The segmentation dataset.
112    """
113    image_paths, gt_paths = get_ocutox_paths(path, download)
114
115    if resize_inputs:
116        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
117        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
118            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
119        )
120
121    return torch_em.default_segmentation_dataset(
122        raw_paths=image_paths,
123        raw_key=None,
124        label_paths=gt_paths,
125        label_key=None,
126        patch_shape=patch_shape,
127        is_seg_dataset=False,
128        **kwargs
129    )
130
131
132def get_ocutox_loader(
133    path: Union[os.PathLike, str],
134    batch_size: int,
135    patch_shape: Tuple[int, int],
136    resize_inputs: bool = False,
137    download: bool = False,
138    **kwargs
139) -> DataLoader:
140    """Get the OCUTOX dataloader for segmentation of ocular toxoplasmosis lesions in fundus images.
141
142    Args:
143        path: Filepath to a folder where the data is downloaded for further processing.
144        batch_size: The batch size for training.
145        patch_shape: The patch shape to use for training.
146        resize_inputs: Whether to resize the inputs to the expected patch shape.
147        download: Whether to download the data if it is not present.
148        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
149
150    Returns:
151        The DataLoader.
152    """
153    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
154    dataset = get_ocutox_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
155    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/records/5156940/files/Ocular_Toxoplasmosis_Data_V3.zip'
CHECKSUM = 'd088dcfe6c678923eed20e18052935cff70a3e0ede8f87f06406caa734727ece'
MASK_SUFFIX_PATTERN = re.compile('-(?:a|i)(?:-\\d+)?$|-\\d+$', re.IGNORECASE)
def get_ocutox_data(path: Union[os.PathLike, str], download: bool = False) -> str:
35def get_ocutox_data(path: Union[os.PathLike, str], download: bool = False) -> str:
36    """Download the OCUTOX dataset.
37
38    Args:
39        path: Filepath to a folder where the data is downloaded for further processing.
40        download: Whether to download the data if it is not present.
41
42    Returns:
43        Filepath where the data is downloaded.
44    """
45    data_dir = os.path.join(path, "images")
46    if os.path.exists(data_dir):
47        return path
48
49    os.makedirs(path, exist_ok=True)
50
51    zip_path = os.path.join(path, "Ocular_Toxoplasmosis_Data_V3.zip")
52    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
53    util.unzip(zip_path=zip_path, dst=path)
54
55    return path

Download the OCUTOX dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_ocutox_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
58def get_ocutox_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
59    """Get paths to the OCUTOX data.
60
61    Args:
62        path: Filepath to a folder where the data is downloaded for further processing.
63        download: Whether to download the data if it is not present.
64
65    Returns:
66        List of filepaths for the image data.
67        List of filepaths for the label data.
68    """
69    data_dir = get_ocutox_data(path=path, download=download)
70
71    image_dir = os.path.join(data_dir, "images")
72    mask_dir = os.path.join(data_dir, "masks")
73
74    image_files = {os.path.basename(p).lower(): p for p in glob(os.path.join(image_dir, "*.*"))}
75    mask_paths = sorted(glob(os.path.join(mask_dir, "*.*")))
76
77    image_paths, matched_mask_paths = [], []
78    for mask_path in mask_paths:
79        fname = os.path.basename(mask_path)
80        stem, ext = os.path.splitext(fname.lower())
81        base_stem = MASK_SUFFIX_PATTERN.sub("", stem)
82        image_path = image_files.get(base_stem + ext)
83        if image_path is None:
84            raise RuntimeError(f"Could not find the matching image for the mask at '{mask_path}'.")
85
86        image_paths.append(image_path)
87        matched_mask_paths.append(mask_path)
88
89    if len(image_paths) == 0 or len(image_paths) != len(matched_mask_paths):
90        raise RuntimeError("Something went wrong with fetching the image and label paths.")
91
92    return image_paths, matched_mask_paths

Get paths to the OCUTOX data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_ocutox_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
 95def get_ocutox_dataset(
 96    path: Union[os.PathLike, str],
 97    patch_shape: Tuple[int, int],
 98    resize_inputs: bool = False,
 99    download: bool = False,
100    **kwargs
101) -> Dataset:
102    """Get the OCUTOX dataset for segmentation of ocular toxoplasmosis lesions in fundus images.
103
104    Args:
105        path: Filepath to a folder where the data is downloaded for further processing.
106        patch_shape: The patch shape to use for training.
107        resize_inputs: Whether to resize the inputs to the expected patch shape.
108        download: Whether to download the data if it is not present.
109        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
110
111    Returns:
112        The segmentation dataset.
113    """
114    image_paths, gt_paths = get_ocutox_paths(path, download)
115
116    if resize_inputs:
117        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
118        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
119            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
120        )
121
122    return torch_em.default_segmentation_dataset(
123        raw_paths=image_paths,
124        raw_key=None,
125        label_paths=gt_paths,
126        label_key=None,
127        patch_shape=patch_shape,
128        is_seg_dataset=False,
129        **kwargs
130    )

Get the OCUTOX dataset for segmentation of ocular toxoplasmosis lesions in fundus images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_ocutox_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
133def get_ocutox_loader(
134    path: Union[os.PathLike, str],
135    batch_size: int,
136    patch_shape: Tuple[int, int],
137    resize_inputs: bool = False,
138    download: bool = False,
139    **kwargs
140) -> DataLoader:
141    """Get the OCUTOX dataloader for segmentation of ocular toxoplasmosis lesions in fundus images.
142
143    Args:
144        path: Filepath to a folder where the data is downloaded for further processing.
145        batch_size: The batch size for training.
146        patch_shape: The patch shape to use for training.
147        resize_inputs: Whether to resize the inputs to the expected patch shape.
148        download: Whether to download the data if it is not present.
149        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
150
151    Returns:
152        The DataLoader.
153    """
154    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
155    dataset = get_ocutox_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
156    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the OCUTOX dataloader for segmentation of ocular toxoplasmosis lesions in fundus images.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.