torch_em.data.datasets.medical.ocutox
The OCUTOX dataset contains annotations for active and inactive lesion segmentation in fundus images of patients with ocular toxoplasmosis.
The dataset contains 412 fundus images collected at the Hospital de Clinicas and the Hospital General Pediatrico Acosta Nu medical centers in Asuncion, Paraguay, of which 280 images (with active and / or inactive toxoplasmosis lesions) have pixel-level lesion masks delineated by ophthalmologists. The remaining images are labeled 'healthy' and do not have a lesion mask.
This dataset is located at https://doi.org/10.5281/zenodo.5156940 (CC BY 4.0). Please cite it if you use this dataset for your research.
1"""The OCUTOX dataset contains annotations for active and inactive lesion segmentation 2in fundus images of patients with ocular toxoplasmosis. 3 4The dataset contains 412 fundus images collected at the Hospital de Clinicas and the 5Hospital General Pediatrico Acosta Nu medical centers in Asuncion, Paraguay, of which 6280 images (with active and / or inactive toxoplasmosis lesions) have pixel-level lesion 7masks delineated by ophthalmologists. The remaining images are labeled 'healthy' and do 8not have a lesion mask. 9 10This dataset is located at https://doi.org/10.5281/zenodo.5156940 (CC BY 4.0). 11Please cite it if you use this dataset for your research. 12""" 13 14import os 15import re 16from glob import glob 17from typing import Union, Tuple, List 18 19from torch.utils.data import Dataset, DataLoader 20 21import torch_em 22 23from .. import util 24 25 26URL = "https://zenodo.org/records/5156940/files/Ocular_Toxoplasmosis_Data_V3.zip" 27CHECKSUM = "d088dcfe6c678923eed20e18052935cff70a3e0ede8f87f06406caa734727ece" 28 29# Masks carry suffixes for lesion sub-regions of the same image, eg. '-a' (active lesion), 30# '-i' (inactive lesion) and numeric variants ('-2', '-3', '-a-2', ...). 31MASK_SUFFIX_PATTERN = re.compile(r"-(?:a|i)(?:-\d+)?$|-\d+$", flags=re.IGNORECASE) 32 33 34def get_ocutox_data(path: Union[os.PathLike, str], download: bool = False) -> str: 35 """Download the OCUTOX dataset. 36 37 Args: 38 path: Filepath to a folder where the data is downloaded for further processing. 39 download: Whether to download the data if it is not present. 40 41 Returns: 42 Filepath where the data is downloaded. 43 """ 44 data_dir = os.path.join(path, "images") 45 if os.path.exists(data_dir): 46 return path 47 48 os.makedirs(path, exist_ok=True) 49 50 zip_path = os.path.join(path, "Ocular_Toxoplasmosis_Data_V3.zip") 51 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 52 util.unzip(zip_path=zip_path, dst=path) 53 54 return path 55 56 57def get_ocutox_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 58 """Get paths to the OCUTOX data. 59 60 Args: 61 path: Filepath to a folder where the data is downloaded for further processing. 62 download: Whether to download the data if it is not present. 63 64 Returns: 65 List of filepaths for the image data. 66 List of filepaths for the label data. 67 """ 68 data_dir = get_ocutox_data(path=path, download=download) 69 70 image_dir = os.path.join(data_dir, "images") 71 mask_dir = os.path.join(data_dir, "masks") 72 73 image_files = {os.path.basename(p).lower(): p for p in glob(os.path.join(image_dir, "*.*"))} 74 mask_paths = sorted(glob(os.path.join(mask_dir, "*.*"))) 75 76 image_paths, matched_mask_paths = [], [] 77 for mask_path in mask_paths: 78 fname = os.path.basename(mask_path) 79 stem, ext = os.path.splitext(fname.lower()) 80 base_stem = MASK_SUFFIX_PATTERN.sub("", stem) 81 image_path = image_files.get(base_stem + ext) 82 if image_path is None: 83 raise RuntimeError(f"Could not find the matching image for the mask at '{mask_path}'.") 84 85 image_paths.append(image_path) 86 matched_mask_paths.append(mask_path) 87 88 if len(image_paths) == 0 or len(image_paths) != len(matched_mask_paths): 89 raise RuntimeError("Something went wrong with fetching the image and label paths.") 90 91 return image_paths, matched_mask_paths 92 93 94def get_ocutox_dataset( 95 path: Union[os.PathLike, str], 96 patch_shape: Tuple[int, int], 97 resize_inputs: bool = False, 98 download: bool = False, 99 **kwargs 100) -> Dataset: 101 """Get the OCUTOX dataset for segmentation of ocular toxoplasmosis lesions in fundus images. 102 103 Args: 104 path: Filepath to a folder where the data is downloaded for further processing. 105 patch_shape: The patch shape to use for training. 106 resize_inputs: Whether to resize the inputs to the expected patch shape. 107 download: Whether to download the data if it is not present. 108 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 109 110 Returns: 111 The segmentation dataset. 112 """ 113 image_paths, gt_paths = get_ocutox_paths(path, download) 114 115 if resize_inputs: 116 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 117 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 118 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 119 ) 120 121 return torch_em.default_segmentation_dataset( 122 raw_paths=image_paths, 123 raw_key=None, 124 label_paths=gt_paths, 125 label_key=None, 126 patch_shape=patch_shape, 127 is_seg_dataset=False, 128 **kwargs 129 ) 130 131 132def get_ocutox_loader( 133 path: Union[os.PathLike, str], 134 batch_size: int, 135 patch_shape: Tuple[int, int], 136 resize_inputs: bool = False, 137 download: bool = False, 138 **kwargs 139) -> DataLoader: 140 """Get the OCUTOX dataloader for segmentation of ocular toxoplasmosis lesions in fundus images. 141 142 Args: 143 path: Filepath to a folder where the data is downloaded for further processing. 144 batch_size: The batch size for training. 145 patch_shape: The patch shape to use for training. 146 resize_inputs: Whether to resize the inputs to the expected patch shape. 147 download: Whether to download the data if it is not present. 148 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 149 150 Returns: 151 The DataLoader. 152 """ 153 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 154 dataset = get_ocutox_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 155 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
35def get_ocutox_data(path: Union[os.PathLike, str], download: bool = False) -> str: 36 """Download the OCUTOX dataset. 37 38 Args: 39 path: Filepath to a folder where the data is downloaded for further processing. 40 download: Whether to download the data if it is not present. 41 42 Returns: 43 Filepath where the data is downloaded. 44 """ 45 data_dir = os.path.join(path, "images") 46 if os.path.exists(data_dir): 47 return path 48 49 os.makedirs(path, exist_ok=True) 50 51 zip_path = os.path.join(path, "Ocular_Toxoplasmosis_Data_V3.zip") 52 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 53 util.unzip(zip_path=zip_path, dst=path) 54 55 return path
Download the OCUTOX dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
58def get_ocutox_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 59 """Get paths to the OCUTOX data. 60 61 Args: 62 path: Filepath to a folder where the data is downloaded for further processing. 63 download: Whether to download the data if it is not present. 64 65 Returns: 66 List of filepaths for the image data. 67 List of filepaths for the label data. 68 """ 69 data_dir = get_ocutox_data(path=path, download=download) 70 71 image_dir = os.path.join(data_dir, "images") 72 mask_dir = os.path.join(data_dir, "masks") 73 74 image_files = {os.path.basename(p).lower(): p for p in glob(os.path.join(image_dir, "*.*"))} 75 mask_paths = sorted(glob(os.path.join(mask_dir, "*.*"))) 76 77 image_paths, matched_mask_paths = [], [] 78 for mask_path in mask_paths: 79 fname = os.path.basename(mask_path) 80 stem, ext = os.path.splitext(fname.lower()) 81 base_stem = MASK_SUFFIX_PATTERN.sub("", stem) 82 image_path = image_files.get(base_stem + ext) 83 if image_path is None: 84 raise RuntimeError(f"Could not find the matching image for the mask at '{mask_path}'.") 85 86 image_paths.append(image_path) 87 matched_mask_paths.append(mask_path) 88 89 if len(image_paths) == 0 or len(image_paths) != len(matched_mask_paths): 90 raise RuntimeError("Something went wrong with fetching the image and label paths.") 91 92 return image_paths, matched_mask_paths
Get paths to the OCUTOX data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
95def get_ocutox_dataset( 96 path: Union[os.PathLike, str], 97 patch_shape: Tuple[int, int], 98 resize_inputs: bool = False, 99 download: bool = False, 100 **kwargs 101) -> Dataset: 102 """Get the OCUTOX dataset for segmentation of ocular toxoplasmosis lesions in fundus images. 103 104 Args: 105 path: Filepath to a folder where the data is downloaded for further processing. 106 patch_shape: The patch shape to use for training. 107 resize_inputs: Whether to resize the inputs to the expected patch shape. 108 download: Whether to download the data if it is not present. 109 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 110 111 Returns: 112 The segmentation dataset. 113 """ 114 image_paths, gt_paths = get_ocutox_paths(path, download) 115 116 if resize_inputs: 117 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True} 118 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 119 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 120 ) 121 122 return torch_em.default_segmentation_dataset( 123 raw_paths=image_paths, 124 raw_key=None, 125 label_paths=gt_paths, 126 label_key=None, 127 patch_shape=patch_shape, 128 is_seg_dataset=False, 129 **kwargs 130 )
Get the OCUTOX dataset for segmentation of ocular toxoplasmosis lesions in fundus images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs to the expected patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
133def get_ocutox_loader( 134 path: Union[os.PathLike, str], 135 batch_size: int, 136 patch_shape: Tuple[int, int], 137 resize_inputs: bool = False, 138 download: bool = False, 139 **kwargs 140) -> DataLoader: 141 """Get the OCUTOX dataloader for segmentation of ocular toxoplasmosis lesions in fundus images. 142 143 Args: 144 path: Filepath to a folder where the data is downloaded for further processing. 145 batch_size: The batch size for training. 146 patch_shape: The patch shape to use for training. 147 resize_inputs: Whether to resize the inputs to the expected patch shape. 148 download: Whether to download the data if it is not present. 149 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 150 151 Returns: 152 The DataLoader. 153 """ 154 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 155 dataset = get_ocutox_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 156 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the OCUTOX dataloader for segmentation of ocular toxoplasmosis lesions in fundus images.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs to the expected patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.